{
  "cases": [
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-001",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "orange",
          "clearance": 2,
          "emergency_permit": true,
          "experience_years": 16,
          "role": "contractor",
          "sponsor": false,
          "suspended": false,
          "training_complete": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        80,
        81
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-002",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "green",
          "clearance": 2,
          "emergency_permit": true,
          "experience_years": 15,
          "role": "staff",
          "sponsor": false,
          "suspended": false,
          "training_complete": false
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        124,
        125
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-003",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 113,
          "active_promotion": false,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": false,
          "member": true,
          "subtotal_credits": 99,
          "weight_kg": 21
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        264,
        265
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-004",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 30,
          "defective": false,
          "final_sale": true,
          "price_credits": 256,
          "recalled": false,
          "receipt": false,
          "unused": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        330,
        331
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-005",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": false,
          "days_since_purchase": 90,
          "defective": true,
          "final_sale": false,
          "price_credits": 22,
          "recalled": true,
          "receipt": false,
          "unused": false
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        312,
        313
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-006",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 1145,
          "active_promotion": true,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": true,
          "member": false,
          "subtotal_credits": 99,
          "weight_kg": 21
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        256,
        257
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-007",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "green",
          "clearance": 2,
          "emergency_permit": false,
          "experience_years": 27,
          "role": "contractor",
          "sponsor": true,
          "suspended": false,
          "training_complete": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        44,
        45
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-008",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 450,
          "active_promotion": false,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": true,
          "member": false,
          "subtotal_credits": 100,
          "weight_kg": 21
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        190,
        191
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-009",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 0,
          "defective": true,
          "final_sale": true,
          "price_credits": 80,
          "recalled": false,
          "receipt": true,
          "unused": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        210,
        211
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-010",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": false,
          "days_since_purchase": 90,
          "defective": false,
          "final_sale": true,
          "price_credits": 368,
          "recalled": false,
          "receipt": true,
          "unused": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        204,
        205
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-011",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "orange",
          "clearance": 3,
          "emergency_permit": true,
          "experience_years": 22,
          "role": "visitor",
          "sponsor": false,
          "suspended": false,
          "training_complete": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        202,
        203
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-012",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "green",
          "clearance": 4,
          "emergency_permit": true,
          "experience_years": 10,
          "role": "contractor",
          "sponsor": false,
          "suspended": false,
          "training_complete": false
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        60,
        61
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-013",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 91,
          "defective": true,
          "final_sale": true,
          "price_credits": 334,
          "recalled": true,
          "receipt": false,
          "unused": false
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        58,
        59
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-014",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 90,
          "defective": true,
          "final_sale": true,
          "price_credits": 141,
          "recalled": false,
          "receipt": false,
          "unused": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        300,
        301
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-015",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 31,
          "defective": false,
          "final_sale": false,
          "price_credits": 247,
          "recalled": true,
          "receipt": true,
          "unused": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        104,
        105
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-016",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 610,
          "active_promotion": false,
          "gift_wrapping": true,
          "hazardous": true,
          "mainland": true,
          "member": false,
          "subtotal_credits": 99,
          "weight_kg": 20
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        4,
        5
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-017",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": false,
          "days_since_purchase": 0,
          "defective": true,
          "final_sale": true,
          "price_credits": 201,
          "recalled": true,
          "receipt": true,
          "unused": false
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        188,
        189
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-018",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": false,
          "days_since_purchase": 30,
          "defective": true,
          "final_sale": false,
          "price_credits": 298,
          "recalled": true,
          "receipt": false,
          "unused": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        208,
        209
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-019",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 856,
          "active_promotion": false,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": true,
          "member": true,
          "subtotal_credits": 100,
          "weight_kg": 19
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        176,
        177
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-020",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "green",
          "clearance": 3,
          "emergency_permit": true,
          "experience_years": 17,
          "role": "visitor",
          "sponsor": false,
          "suspended": true,
          "training_complete": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        302,
        303
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-021",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": false,
          "badge_color": "blue",
          "clearance": 3,
          "emergency_permit": false,
          "experience_years": 19,
          "role": "visitor",
          "sponsor": false,
          "suspended": true,
          "training_complete": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        14,
        15
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-022",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "orange",
          "clearance": 1,
          "emergency_permit": true,
          "experience_years": 29,
          "role": "staff",
          "sponsor": false,
          "suspended": false,
          "training_complete": false
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        318,
        319
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-023",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": false,
          "days_since_purchase": 0,
          "defective": false,
          "final_sale": false,
          "price_credits": 247,
          "recalled": true,
          "receipt": false,
          "unused": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        152,
        153
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-024",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": false,
          "days_since_purchase": 120,
          "defective": true,
          "final_sale": false,
          "price_credits": 73,
          "recalled": false,
          "receipt": true,
          "unused": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        348,
        349
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-025",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 320,
          "active_promotion": true,
          "gift_wrapping": false,
          "hazardous": true,
          "mainland": false,
          "member": true,
          "subtotal_credits": 101,
          "weight_kg": 21
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        62,
        63
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-026",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "orange",
          "clearance": 1,
          "emergency_permit": true,
          "experience_years": 16,
          "role": "visitor",
          "sponsor": true,
          "suspended": false,
          "training_complete": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        246,
        247
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-027",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 0,
          "defective": true,
          "final_sale": true,
          "price_credits": 9,
          "recalled": false,
          "receipt": true,
          "unused": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        102,
        103
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-028",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": false,
          "badge_color": "orange",
          "clearance": 3,
          "emergency_permit": true,
          "experience_years": 22,
          "role": "visitor",
          "sponsor": true,
          "suspended": false,
          "training_complete": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        82,
        83
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-029",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 2564,
          "active_promotion": true,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": false,
          "member": true,
          "subtotal_credits": 99,
          "weight_kg": 19
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        308,
        309
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-030",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "green",
          "clearance": 3,
          "emergency_permit": true,
          "experience_years": 25,
          "role": "contractor",
          "sponsor": true,
          "suspended": true,
          "training_complete": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        314,
        315
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-031",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 30,
          "defective": false,
          "final_sale": false,
          "price_credits": 290,
          "recalled": true,
          "receipt": true,
          "unused": false
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        242,
        243
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-032",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 1676,
          "active_promotion": false,
          "gift_wrapping": false,
          "hazardous": false,
          "mainland": true,
          "member": false,
          "subtotal_credits": 100,
          "weight_kg": 20
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        186,
        187
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-033",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 2604,
          "active_promotion": true,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": true,
          "member": false,
          "subtotal_credits": 101,
          "weight_kg": 21
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        306,
        307
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-034",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "green",
          "clearance": 1,
          "emergency_permit": true,
          "experience_years": 0,
          "role": "visitor",
          "sponsor": false,
          "suspended": false,
          "training_complete": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        146,
        147
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-035",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": false,
          "days_since_purchase": 120,
          "defective": true,
          "final_sale": true,
          "price_credits": 308,
          "recalled": false,
          "receipt": true,
          "unused": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        240,
        241
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-036",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": false,
          "days_since_purchase": 91,
          "defective": true,
          "final_sale": false,
          "price_credits": 31,
          "recalled": false,
          "receipt": true,
          "unused": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        74,
        75
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-037",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "green",
          "clearance": 3,
          "emergency_permit": true,
          "experience_years": 0,
          "role": "contractor",
          "sponsor": true,
          "suspended": false,
          "training_complete": false
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        358,
        359
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-038",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": false,
          "badge_color": "green",
          "clearance": 1,
          "emergency_permit": false,
          "experience_years": 19,
          "role": "visitor",
          "sponsor": false,
          "suspended": true,
          "training_complete": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        260,
        261
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-039",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 1700,
          "active_promotion": true,
          "gift_wrapping": false,
          "hazardous": false,
          "mainland": true,
          "member": true,
          "subtotal_credits": 49,
          "weight_kg": 19
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        72,
        73
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-040",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 2869,
          "active_promotion": false,
          "gift_wrapping": false,
          "hazardous": false,
          "mainland": true,
          "member": true,
          "subtotal_credits": 50,
          "weight_kg": 19
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        78,
        79
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-041",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": false,
          "badge_color": "orange",
          "clearance": 2,
          "emergency_permit": true,
          "experience_years": 8,
          "role": "visitor",
          "sponsor": false,
          "suspended": false,
          "training_complete": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        66,
        67
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-042",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 1767,
          "active_promotion": false,
          "gift_wrapping": false,
          "hazardous": false,
          "mainland": true,
          "member": true,
          "subtotal_credits": 99,
          "weight_kg": 20
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        362,
        363
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-043",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 31,
          "defective": true,
          "final_sale": true,
          "price_credits": 214,
          "recalled": true,
          "receipt": true,
          "unused": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        128,
        129
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-044",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 916,
          "active_promotion": true,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": true,
          "member": true,
          "subtotal_credits": 101,
          "weight_kg": 20
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        238,
        239
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-045",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 1675,
          "active_promotion": true,
          "gift_wrapping": true,
          "hazardous": true,
          "mainland": false,
          "member": true,
          "subtotal_credits": 50,
          "weight_kg": 21
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        254,
        255
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-046",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": false,
          "days_since_purchase": 29,
          "defective": false,
          "final_sale": false,
          "price_credits": 208,
          "recalled": false,
          "receipt": false,
          "unused": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        120,
        121
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-047",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "orange",
          "clearance": 1,
          "emergency_permit": false,
          "experience_years": 29,
          "role": "visitor",
          "sponsor": true,
          "suspended": true,
          "training_complete": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        64,
        65
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-048",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": false,
          "badge_color": "blue",
          "clearance": 3,
          "emergency_permit": true,
          "experience_years": 25,
          "role": "staff",
          "sponsor": false,
          "suspended": true,
          "training_complete": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        278,
        279
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-049",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 0,
          "defective": true,
          "final_sale": false,
          "price_credits": 222,
          "recalled": true,
          "receipt": true,
          "unused": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        144,
        145
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-050",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 1795,
          "active_promotion": false,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": true,
          "member": true,
          "subtotal_credits": 101,
          "weight_kg": 20
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        274,
        275
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-051",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 90,
          "defective": true,
          "final_sale": true,
          "price_credits": 171,
          "recalled": false,
          "receipt": true,
          "unused": true
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        168,
        169
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-052",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "green",
          "clearance": 3,
          "emergency_permit": true,
          "experience_years": 27,
          "role": "staff",
          "sponsor": true,
          "suspended": false,
          "training_complete": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        174,
        175
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-053",
      "question": "Is this item eligible for a refund under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "refund",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
        "record": {
          "customer_member": true,
          "days_since_purchase": 120,
          "defective": false,
          "final_sale": false,
          "price_credits": 28,
          "recalled": true,
          "receipt": true,
          "unused": true
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        226,
        227
      ],
      "audit": null,
      "family": "refund"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-054",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 2681,
          "active_promotion": false,
          "gift_wrapping": false,
          "hazardous": false,
          "mainland": true,
          "member": true,
          "subtotal_credits": 51,
          "weight_kg": 19
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        326,
        327
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-055",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 2316,
          "active_promotion": true,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": true,
          "member": false,
          "subtotal_credits": 49,
          "weight_kg": 19
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        310,
        311
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-056",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "green",
          "clearance": 3,
          "emergency_permit": false,
          "experience_years": 8,
          "role": "contractor",
          "sponsor": true,
          "suspended": false,
          "training_complete": false
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        126,
        127
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-057",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 1501,
          "active_promotion": false,
          "gift_wrapping": true,
          "hazardous": false,
          "mainland": false,
          "member": false,
          "subtotal_credits": 101,
          "weight_kg": 20
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        158,
        159
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-058",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 2519,
          "active_promotion": true,
          "gift_wrapping": false,
          "hazardous": true,
          "mainland": true,
          "member": false,
          "subtotal_credits": 0,
          "weight_kg": 20
        }
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        224,
        225
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-059",
      "question": "Is this person allowed laboratory access under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "access",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
        "record": {
          "account_active": true,
          "badge_color": "orange",
          "clearance": 3,
          "emergency_permit": true,
          "experience_years": 13,
          "role": "contractor",
          "sponsor": false,
          "suspended": false,
          "training_complete": false
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        270,
        271
      ],
      "audit": null,
      "family": "access"
    },
    {
      "criteria": {
        "false": "The answer to the stated question is no.",
        "true": "The answer to the stated question is yes."
      },
      "experiment": "policy",
      "id": "policy-060",
      "question": "Does this order qualify for free delivery under the supplied policy?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_programmatic",
        "rule_family": "delivery",
        "seed": 20261002,
        "validation": "Exact boolean rule over supplied record fields"
      },
      "state": {
        "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
        "record": {
          "account_age_days": 1863,
          "active_promotion": true,
          "gift_wrapping": false,
          "hazardous": false,
          "mainland": true,
          "member": true,
          "subtotal_credits": 100,
          "weight_kg": 20
        }
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        166,
        167
      ],
      "audit": null,
      "family": "delivery"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0659",
      "question": "does the host country for the world cup get an automatic bid?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 659,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The hosts of the World Cup receive an automatic berth. Unlike many other sports, results of the previous World Cups or of the continental championships are not taken into account. Until 2002, the defending champions also received an automatic berth, but starting from the 2006 World Cup this is no longer the case."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        212,
        213
      ],
      "audit": {
        "id": "boolq-dev-0659",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly gives the host an automatic berth."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2774",
      "question": "are walgreens and rite aid owned by the same company?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2774,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Rite Aid began in 1962, opening its first store in Scranton, Pennsylvania; it was called Thrift D Discount Center. After several years of growth, Rite Aid adopted its current name and debuted as a public company in 1968. As of 2017, Rite Aid is publicly traded on the New York Stock Exchange under the symbol RAD. Its major competitors are CVS and Walgreens. In late 2015, Walgreens announced that it would acquire Rite Aid for $9.4 billion pending approval. However, on June 29, 2017, over fear of antitrust regulations, Walgreens Boots Alliance announced it would buy roughly half of Rite Aid's stores for $5.18 billion. On September 19, 2017, the Federal Trade Commission (FTC) approved a fourth deal agreement to purchase Rite Aid with 1,932 stores for $4.38 billion total."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        286,
        287
      ],
      "audit": {
        "id": "boolq-dev-2774",
        "label": "no",
        "category": "ambiguous",
        "evidence": "Source describes a proposed acquisition followed by acquisition of stores; its last sentence can be read as acquisition of Rite Aid itself. Company ownership versus store ownership and the 2017 time frame need care."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0579",
      "question": "do female and male blue jays look the same?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 579,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Its plumage is lavender-blue to mid-blue in the crest, back, wings, and tail, and its face is white. The underside is off-white and the neck is collared with black which extends to the sides of the head. The wing primaries and tail are strongly barred with black, sky-blue and white. The bill, legs, and eyes are all black. Males and females are almost identical, but the male is slightly larger."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        10,
        11
      ],
      "audit": {
        "id": "boolq-dev-0579",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage says males and females are almost identical, while acknowledging male size differs."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2217",
      "question": "are swift codes and bsb numbers the same?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2217,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "A Bank State Branch (often referred to as ``BSB'') is the name used in Australia for a bank code, which is a branch identifier. The BSB is normally used in association with the account number system used by each financial institution. The structure of the BSB + account number does not permit for account numbers to be transferable between financial institutions. While similar in structure, the New Zealand and Australian systems are only used in domestic transactions and are incompatible with each other. For international transfers, a SWIFT code is used in addition to the BSB and account number."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        178,
        179
      ],
      "audit": {
        "id": "boolq-dev-2217",
        "label": "no",
        "category": "supported",
        "evidence": "SWIFT is used in addition to BSB for international transfers; they are distinct identifiers."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1517",
      "question": "are all world cup matches played in russia?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1517,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The 2018 FIFA World Cup was the 21st FIFA World Cup, an international football tournament contested by the men's national teams of the member associations of FIFA once every four years. It took place in Russia from 14 June to 15 July 2018. It was the first World Cup to be held in Eastern Europe, and the 11th time that it had been held in Europe. At an estimated cost of over $14.2 billion, it was the most expensive World Cup. It was also the first World Cup to use the video assistant referee (VAR) system."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        134,
        135
      ],
      "audit": {
        "id": "boolq-dev-1517",
        "label": "no",
        "category": "ambiguous",
        "evidence": "Passage says the 2018 tournament took place in Russia. No is defensible for all World Cups across history, while yes is natural for all matches in the 2018 tournament. The question does not identify its time scope."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1662",
      "question": "can you get turf toe on other toes?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1662,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Turf toe is named from the injury being associated with playing sports on rigid surfaces such as artificial turf and is a fairly common injury among professional American football players. Often, the injury occurs when someone or something falls on the back of the calf while that leg's knee and tips of the toes are touching the ground. The toe is hyperextended and thus the joint is injured. Additionally, athletic shoes with very flexible soles combined with cleats that ``grab'' the turf will cause overextension of the big toe. This can occur on the lesser toes as well. It has also been observed in sports beyond American football, including soccer, basketball, rugby, volleyball, and tae kwon do. This is a primary reason why many athletes prefer natural grass to turf, because it is softer."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        262,
        263
      ],
      "audit": {
        "id": "boolq-dev-1662",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly says the injury can occur on lesser toes."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2868",
      "question": "do you need a permit to open carry in nc?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2868,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Open carry is also legal throughout North Carolina. In the town of Chapel Hill, open carry is restricted to guns of a certain minimum size, under the theory that small, concealable handguns are more often associated with criminal activity. No permit is required to carry a handgun openly in North Carolina. In the court case of State v. Kerner(1921) the defendant ended up getting into some type of confrontation with another man. The defendant proceeded to walk back to his place of work, get his gun, and then return to the scene to fight. The defendant ended up being charged with ``carrying a concealed weapon'' and ``carrying his pistol off his premises unconcealed,'' which violated a local act applicable to Forsyth County and ended up being a misdemeanor. The defendant was taken to trial and the trial judge then dismissed the charge as unconstitutional. The state then appealed, and the supreme court affirmed. During court, the court stated at the beginning that the Second Amendment did not apply, because ``the first ten amendments to the United States Constitution are restrictions on the federal authority and not the states.'' Therefore, with that being said, it focused more on the state constitution. The state constitution states that: ``A well regulated militia, being necessary to the security of a free state, the right of the people to keep and bear arms, shall not be infringed.'' The court viewed the provision as protecting the right to carry arms in public. Forsyth County's local act was condemned and seen as distasteful, because it ended up putting a restriction on a persons right to carry a pistol, more so an unconcealed pistol. Although, the case of State v. Kerner helped/made more clear the allowance of openly carrying a pistol, it does not preclude all regulations regarding the carrying of firearms."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        292,
        293
      ],
      "audit": {
        "id": "boolq-dev-2868",
        "label": "no",
        "category": "time_or_scope",
        "evidence": "Passage explicitly says no permit is required for open handgun carry in North Carolina. This is passage-era legal text; no current-law verification was performed."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1447",
      "question": "can an image be formed without a screen?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1447,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "In optics, a virtual image is an image formed when the outgoing rays from a point on an object always diverge. The image appears to be located at the point of apparent divergence. Because the rays never really converge, a virtual image cannot be projected onto a screen. In diagrams of optical systems, virtual rays are conventionally represented by dotted lines. Virtual images are located by tracing the real rays that emerge from an optical device (lens, mirror, or some combination) backward to a perceived point of origin."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        32,
        33
      ],
      "audit": {
        "id": "boolq-dev-1447",
        "label": "yes",
        "category": "supported",
        "evidence": "A virtual image is described as formed yet unable to be projected onto a screen."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0383",
      "question": "do you qualify for the world cup if you host it?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 383,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The hosts of the World Cup receive an automatic berth. Unlike many other sports, results of the previous World Cups or of the continental championships are not taken into account. Until 2002, the defending champions also received an automatic berth, but starting from the 2006 World Cup this is no longer the case."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        332,
        333
      ],
      "audit": {
        "id": "boolq-dev-0383",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly gives the host an automatic berth. It repeats the passage in case 0659 with a paraphrased question."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2665",
      "question": "is there a place called cabot cove maine?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2665,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The show revolves around the day-to-day life of Jessica Fletcher, a childless, widowed, retired English teacher who becomes a successful mystery writer. Despite fame and fortune, Jessica remains a resident of Cabot Cove, a small coastal community in Maine, and maintains her links with all of her old friends, never letting her success go to her head. Exterior shots of Cabot Cove were filmed in Mendocino, California. The fictional ``Cabot Cove'' name for the series' coastal town was derived from the name of an actual bay harbor inlet in Kennebunkport, Maine, located near the town's center, on the road where motels and lobster shack dives are located."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        340,
        341
      ],
      "audit": {
        "id": "boolq-dev-2665",
        "label": "no",
        "category": "supported",
        "evidence": "The town name is explicitly described as fictional; the real place mentioned is a different bay/inlet."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2041",
      "question": "are all public companies listed on the stock exchange?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2041,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "An unlisted public company is a public company that is not listed on any stock exchange. Though the criteria vary somewhat between jurisdictions, a public company is a company that is registered as such and generally has a minimum share capital and a minimum number of shareholders. Each stock exchange has its own listing requirements which a company (or other entity) wishing to be listed must meet. Besides not qualifying to be listed, a public company may choose not to be listed on a stock exchange for a number of reasons, including because it is too small to qualify for a stock exchange listing, does not seek public investors, or there are too few shareholders for a listing. There is a cost to the listed entities, in the listing process and ongoing costs as well as in compliance costs such as the maintenance of a company register."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        154,
        155
      ],
      "audit": {
        "id": "boolq-dev-2041",
        "label": "no",
        "category": "supported",
        "evidence": "The passage defines an unlisted public company as a public company not listed on any exchange."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1296",
      "question": "are the jungle and rainforest the same thing?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1296,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "A jungle is land covered with dense vegetation dominated by trees. Application of the term has varied greatly during the past recent centuries. Before the 1970s, tropical rainforests were generally referred to as jungles but this terminology has fallen out of usage. Jungles in Western literature can represent a less civilised or unruly space outside the control of civilisation, attributed to the jungle's association in colonial discourse with places colonised by Europeans."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        236,
        237
      ],
      "audit": {
        "id": "boolq-dev-1296",
        "label": "no",
        "category": "ambiguous",
        "evidence": "Passage describes the history of rainforest being called jungle and a broad definition of jungle; it does not directly settle equivalence. The no label is plausible under a strict distinction, but the passage also supports colloquial overlap."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1114",
      "question": "is a timber wolf the same as a gray wolf?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1114,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The gray wolf (Canis lupus), also known as the timber wolf, western wolf, or simply, wolf, is a canine native to the wilderness and remote areas of Eurasia and North America. It is the largest extant member of its family, with males averaging 43--45 kg (95--99 lb) and females 36--38.5 kg (79--85 lb). Like the red wolf, it is distinguished from other Canis species by its larger size and less pointed features, particularly on the ears and muzzle. Its winter fur is long and bushy and predominantly a mottled gray in color, although nearly pure white, red, and brown to black also occur. Mammal Species of the World (3rd ed., 2005), a standard reference work in zoology, recognises 38 subspecies of C. lupus.."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        16,
        17
      ],
      "audit": {
        "id": "boolq-dev-1114",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly calls gray wolf also known as timber wolf."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1953",
      "question": "is all of australia in the same time zone?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1953,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Australia uses three main time zones: Australian Western Standard Time (AWST; UTC+08:00), Australian Central Standard Time (ACST; UTC+09:30), and Australian Eastern Standard Time (AEST; UTC+10:00). Time is regulated by the individual state governments, some of which observe daylight saving time (DST). Australia's external territories observe different time zones."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        324,
        325
      ],
      "audit": {
        "id": "boolq-dev-1953",
        "label": "no",
        "category": "supported",
        "evidence": "Passage lists three main Australian time zones and different external-territory zones."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2493",
      "question": "can you make a citizen's arrest in the uk?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2493,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Despite the practice's name, in most countries, the arresting person is usually designated as a person with arrest powers, who need not be a citizen of the country in which they are acting. For example, in the British jurisdiction of England and Wales, the power comes from section 24A(2) of the Police and Criminal Evidence Act 1984, called ``any person arrest''. This legislation states ``any person'' has these powers, and does not state that they need to be a British citizen."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        40,
        41
      ],
      "audit": {
        "id": "boolq-dev-2493",
        "label": "yes",
        "category": "time_or_scope",
        "evidence": "Passage supports any-person arrest in England and Wales. That supports an existential reading of in the UK, but does not establish every UK jurisdiction or current eligibility conditions."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-3009",
      "question": "does the subway in nyc run 24 hours?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 3009,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The New York City Subway is a rapid transit system owned by the City of New York and leased to the New York City Transit Authority, a subsidiary agency of the state-run Metropolitan Transportation Authority (MTA). Opened in 1904, the New York City Subway is one of the world's oldest public transit systems, one of the world's most used metro systems, and the metro system with the most stations. It offers service 24 hours per day on every day of the year, though some routes may operate only part-time."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        86,
        87
      ],
      "audit": {
        "id": "boolq-dev-3009",
        "label": "yes",
        "category": "time_or_scope",
        "evidence": "Passage explicitly says service is offered 24 hours daily, while some routes are part-time. It is a source-era system-wide statement, not a guarantee for every route or current service."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2504",
      "question": "did the actor who played gandalf play dumbledore?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2504,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Ian McKellen was offered the role, but he turned it down, having played the similar character Gandalf in The Lord of the Rings trilogy, as well as feeling it would have been inappropriate to take Harris's role, as Harris had called McKellen a ``dreadful'' actor. Harris's family had expressed an interest in seeing Peter O'Toole being chosen as his replacement."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        52,
        53
      ],
      "audit": {
        "id": "boolq-dev-2504",
        "label": "no",
        "category": "supported",
        "evidence": "Passage says Ian McKellen was offered the role and turned it down, while identifying him as Gandalf. The intended role is supplied by the question."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-3125",
      "question": "is there such a thing as an upside down rainbow?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 3125,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The circumzenithal arc, also called the circumzenith arc (CZA), upside-down rainbow, and the Bravais arc, is an optical phenomenon similar in appearance to a rainbow, but belonging to the family of halos arising from refraction of sunlight through ice crystals, generally in cirrus or cirrostratus clouds, rather than from raindrops. The arc is located at a considerable distance (approximately 46°) above the sun and at most forms a quarter of a circle centered on the zenith. It has been called ``a smile in the sky'', its first impression being that of an upside-down rainbow. The CZA is one of the brightest and most colorful members of the halo family. Its colors, ranging from violet on top to red at the bottom, are purer than those of a rainbow because there is much less overlap in their formation."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        116,
        117
      ],
      "audit": {
        "id": "boolq-dev-3125",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage identifies an optical phenomenon called an upside-down rainbow, while distinguishing it physically from rainbows."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1092",
      "question": "is new york in the new england region?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1092,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "New England is a geographical region comprising six states of the northeastern United States: Maine, Vermont, New Hampshire, Massachusetts, Rhode Island, and Connecticut. It is bordered by the state of New York to the west and by the Canadian provinces of New Brunswick and Quebec to the northeast and north, respectively. The Atlantic Ocean is to the east and southeast, and Long Island Sound is to the south. Boston is New England's largest city as well as the capital of Massachusetts. The largest metropolitan area is Greater Boston, which also includes Worcester, Massachusetts (the second-largest city in New England), Manchester, New Hampshire (the largest city in New Hampshire), and Providence, Rhode Island (the capital and largest city of Rhode Island), with nearly a third of the entire region's population."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        150,
        151
      ],
      "audit": {
        "id": "boolq-dev-1092",
        "label": "no",
        "category": "supported",
        "evidence": "Passage lists the six New England states and describes New York as a bordering state."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0055",
      "question": "does damon and elena get together in season 3?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 55,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "In the third season, Damon helps Elena in bringing his brother, Stefan, back to Mystic Falls after Stefan becomes Klaus' henchman. The arrangement transpired after a bargain for his blood that would cure Damon of the werewolf bite he had received from Tyler. At first, he is reluctant to involve Elena in the rescue attempts, employing Alaric Saltzman, Elena's guardian, instead as Klaus does not know that Elena is alive after the sacrifice which frees Klaus' hybrid side. However, Elena involves herself, desperate to find Stefan. Damon, though hesitant at first, is unable to refuse her because of his love for her. He also points out to her that she once turned back from finding Stefan since she knew Damon would be in danger, clearly showing that she also has feelings for him. He tells her that ``when (he) drag(s) (his) brother from the edge to deliver him back to (her), (he) wants her to remember the things (she) felt while he was gone.'' When Stefan finally returns to Mystic Falls, his attitude is different from that of the first and second seasons. This causes a rift between Elena and Stefan whereas the relationship between Damon and Elena becomes closer and more intimate. A still loyal Elena, however, refuses to admit her feelings for Damon. In 'Dangerous Liaisons', Elena, frustrated with her feelings for him, tells Damon that his love for her may be a problem, and that this could be causing all their troubles. This incenses Damon, causing him to revert to the uncaring and reckless Damon seen in the previous seasons. The rocky relationship between the two continues until the sexual tension hits the fan and in a moment of heated passion, Elena -- for the first time in the three seasons -- kisses Damon of her own accord. This kiss finally causes Elena to admit that she loves both brothers and realize that she must ultimately make her choice as her own ancestress, Katherine Pierce, who turned the brothers, once did. In assessment of her feelings for Damon, she states this: ``Damon just sort of snuck up on me. He got under my skin and no matter what I do, I can't shake him.'' In the season finale, a trip designed to get her to safety forces Elena to make her choice: to go to Damon and possibly see him one last time; or to go to Stefan and her friends and see them one last time. She chooses the latter when she calls Damon to tell him her decision. Damon, who is trying to stop Alaric, accepts what she says and she tells him that maybe if she had met Damon before she had met Stefan, her choice may have been different. This statement causes Damon to remember the first night he did meet Elena which was, in fact, the night her parents died - before she had met Stefan. Not wanting anyone to know he was in town and after giving her some advice about life and love, Damon compels her to forget. He remembers this as he fights Alaric and seems accepting of his death when Alaric, whose life line is tied to Elena's, suddenly collapses in his arms. Damon is grief-stricken, knowing that this means that Elena has also died and yells, ``No! You are not dead!'' A heartbroken Damon then goes to the hospital demanding to see Elena when the doctor, Meredith Fell, tells him that she gave Elena vampire blood. The last shot of the season finale episode shows Elena in transition."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        50,
        51
      ],
      "audit": {
        "id": "boolq-dev-0055",
        "label": "no",
        "category": "ambiguous",
        "evidence": "Passage describes kissing and mutual feelings in season 3 but a final choice of Stefan. Get together could mean a romantic encounter or becoming a couple; the no label uses the latter sense."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2791",
      "question": "is new york city and manhattan the same thing?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2791,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "New York City is split up into five boroughs, which are the Bronx, Brooklyn, Manhattan, Queens, and Staten Island. Each borough has the same boundaries as a county of the state. The county governments were dissolved when the city consolidated in 1898, along with all city, town, and village governments within each county. The term borough was adopted to describe a unique form of governmental administration for each of the five fundamental constituent parts of the newly consolidated city."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        34,
        35
      ],
      "audit": {
        "id": "boolq-dev-2791",
        "label": "no",
        "category": "supported",
        "evidence": "Manhattan is listed as one of five constituent boroughs of New York City."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1842",
      "question": "is carbon a metal on the periodic table?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1842,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Carbon (from Latin: carbo ``coal'') is a chemical element with symbol C and atomic number 6. It is nonmetallic and tetravalent--making four electrons available to form covalent chemical bonds. It belongs to group 14 of the periodic table. Three isotopes occur naturally, C and C being stable, while C is a radionuclide, decaying with a half-life of about 5,730 years. Carbon is one of the few elements known since antiquity."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        342,
        343
      ],
      "audit": {
        "id": "boolq-dev-1842",
        "label": "yes",
        "category": "contradiction",
        "evidence": "Published label is yes, but the passage explicitly describes carbon as nonmetallic. Under the ordinary meaning of the question, passage-supported answer is no."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2951",
      "question": "can the word actor be used for females?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2951,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Within the profession, the re-adoption of the neutral term dates to the post-war period of the 1950 and '60s, when the contributions of women to cultural life in general were being reviewed. When The Observer and The Guardian published their new joint style guide in 2010, it stated ``Use (''actor``) for both male and female actors; do not use actress except when in name of award, e.g. Oscar for best actress.'' The guide's authors stated that ``actress comes into the same category as authoress, comedienne, manageress, 'lady doctor', 'male nurse' and similar obsolete terms that date from a time when professions were largely the preserve of one sex (usually men).'' (See male as norm). ``As Whoopi Goldberg put it in an interview with the paper: 'An actress can only play a woman. I'm an actor -- I can play anything.''' The UK performers' union Equity has no policy on the use of ``actor'' or ``actress''. An Equity spokesperson said that the union does not believe that there is a consensus on the matter and stated that the ``...subject divides the profession.'' In 2009, the Los Angeles Times stated that ``Actress'' remains the common term used in major acting awards given to female recipients (e.g., Academy Award for Best Actress)."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        216,
        217
      ],
      "audit": {
        "id": "boolq-dev-2951",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage gives explicit style guidance allowing actor for both women and men."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0852",
      "question": "can a girl scout be an eagle scout?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 852,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The Gold Award is often compared to the Eagle Scout in the Boy Scouts of America."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        290,
        291
      ],
      "audit": {
        "id": "boolq-dev-0852",
        "label": "no",
        "category": "ambiguous",
        "evidence": "Only passage: the Gold Award is often compared to Eagle Scout. This does not state eligibility rules, whether a person can join both organizations, or the relevant date. It cannot establish the no label by itself."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1171",
      "question": "is turkey a part of the european union?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1171,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Relations between the European Union (EU) and Turkey were established in 1959 and the institutional framework is shaped formally since 1963 Ankara Agreement. Turkey is one of the EU's main partners in the Middle East and both are members of the European Union--Turkey Customs Union. The EU and Turkey have a common land border through the EU member states Bulgaria and Greece."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        98,
        99
      ],
      "audit": {
        "id": "boolq-dev-1171",
        "label": "no",
        "category": "time_or_scope",
        "evidence": "Passage treats Turkey and the EU as separate partners with a common border and customs union; the no label is consistent with that source-era reading. Present membership was not checked."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1550",
      "question": "is australia part of the ring of fire?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1550,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The Ring of Fire is a direct result of plate tectonics: the movement and collisions of lithospheric plates. The eastern section of the ring is the result of the Nazca Plate and the Cocos Plate being subducted beneath the westward-moving South American Plate. The Cocos Plate is being subducted beneath the Caribbean Plate, in Central America. A portion of the Pacific Plate and the small Juan de Fuca Plate are being subducted beneath the North American Plate. Along the northern portion, the northwestward-moving Pacific plate is being subducted beneath the Aleutian Islands arc. Farther west, the Pacific plate is being subducted along the Kamchatka Peninsula arcs on south past Japan. The southern portion is more complex, with a number of smaller tectonic plates in collision with the Pacific plate from the Mariana Islands, the Philippines, Bougainville, Tonga, and New Zealand; this portion excludes Australia, since it lies in the center of its tectonic plate. Indonesia lies between the Ring of Fire along the northeastern islands adjacent to and including New Guinea and the Alpide belt along the south and west from Sumatra, Java, Bali, Flores, and Timor."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        8,
        9
      ],
      "audit": {
        "id": "boolq-dev-1550",
        "label": "no",
        "category": "supported",
        "evidence": "Passage explicitly says the relevant part of the Ring of Fire excludes Australia."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2116",
      "question": "is father's day celebrated in the netherlands?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2116,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "In the Netherlands, Father's Day (Vaderdag) is celebrated on the third Sunday of June and is not a public holiday. Traditionally, as on Mother's Day, fathers get breakfast in bed made by their children and families gather together and have dinner, usually at the grandparents' house. In recent years, families also started having dinner out, and as on Mother's Day, it is one of the busiest days for restaurants. At school, children handcraft their present for their fathers. Consumer goods companies have all sorts of special offers for fathers: socks, ties, electronics, suits, and men's healthcare products."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        304,
        305
      ],
      "audit": {
        "id": "boolq-dev-2116",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly states that Father's Day is celebrated in the Netherlands."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0341",
      "question": "is the speed of sound constant for all materials?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 341,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "In common everyday speech, speed of sound refers to the speed of sound waves in air. However, the speed of sound varies from substance to substance: sound travels most slowly in gases; it travels faster in liquids; and faster still in solids. For example, (as noted above), sound travels at 343 m/s in air; it travels at 1,484 m/s in water (4.3 times as fast as in air); and at 5,120 m/s in iron (about 15 times as fast as in air). In an exceptionally stiff material such as diamond, sound travels at 12,000 metres per second (26,843 mph); (about 35 times as fast as in air) which is around the maximum speed that sound will travel under normal conditions."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        88,
        89
      ],
      "audit": {
        "id": "boolq-dev-0341",
        "label": "no",
        "category": "supported",
        "evidence": "Passage explicitly says sound speed varies by substance and provides different values."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2098",
      "question": "is a brother in law considered a relative?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2098,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The immediate family is a defined group of relations, used in rules or laws to determine which members of a person's family are affected by those rules. It normally includes a person's parents, spouses, siblings, children, or an individual related by blood whose close association is an equivalent of a family relationship. It can contain others connected by birth, adoption, marriage, civil partnership, or cohabitation, such as grandparents, great-grandparents, grandchildren, great-grandchildren, aunts, uncles, siblings-in-law, half-siblings, cousins, adopted children and step-parents/step-children, and cohabiting partners."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        118,
        119
      ],
      "audit": {
        "id": "boolq-dev-2098",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage includes siblings-in-law among relations connected by marriage."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2597",
      "question": "is hellmann's and best foods the same company?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2597,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Hellmann's and Best Foods are brand names that are used for the same line of mayonnaise and other food products. The Hellmann's brand is sold in the United States east of the Rocky Mountains, and also in Latin America, Europe, Australia, the Middle East, Canada and South Africa. The Best Foods brand is sold in the United States west of the Rocky Mountains, and also in Asia, Australia, and New Zealand."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        266,
        267
      ],
      "audit": {
        "id": "boolq-dev-2597",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage identifies Hellmann's and Best Foods as names for the same product line. The yes label follows the intended common-brand reading, rather than a separately verified legal-entity claim."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0468",
      "question": "is there a floating mass of garbage in the ocean?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 468,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The Great Pacific garbage patch, also described as the Pacific trash vortex, is a gyre of marine debris particles in the central North Pacific Ocean discovered between 1985 and 1988. It is located roughly between 135°W to 155°W and 35°N to 42°N. The collection of plastic, floating trash halfway between Hawaii and California extends over an indeterminate area of widely varying range depending on the degree of plastic concentration used to define the affected area."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        350,
        351
      ],
      "audit": {
        "id": "boolq-dev-0468",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage describes floating plastic and debris in the North Pacific; mass is read as a collection, not a solid island."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0065",
      "question": "do all triangles have at least two acute angles?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 65,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "An acute triangle is a triangle with all three angles acute (less than 90°). An obtuse triangle is one with one obtuse angle (greater than 90°) and two acute angles. Since a triangle's angles must sum to 180°, no triangle can have more than one obtuse angle."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        192,
        193
      ],
      "audit": {
        "id": "boolq-dev-0065",
        "label": "yes",
        "category": "supported",
        "evidence": "The angle-sum statement and acute/obtuse definitions support at least two acute angles for ordinary nondegenerate Euclidean triangles."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-3114",
      "question": "is a midsummer night's dream a comedy?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 3114,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "A Midsummer Night's Dream is a comedy written by William Shakespeare in 1595/96. It portrays the events surrounding the marriage of Theseus, the Duke of Athens, to Hippolyta, the former queen of the Amazons. These include the adventures of four young Athenian lovers and a group of six amateur actors (the mechanicals) who are controlled and manipulated by the fairies who inhabit the forest in which most of the play is set. The play is one of Shakespeare's most popular works for the stage and is widely performed across the world."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        18,
        19
      ],
      "audit": {
        "id": "boolq-dev-3114",
        "label": "yes",
        "category": "supported",
        "evidence": "The opening sentence explicitly calls the play a comedy."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0877",
      "question": "is the national government the same as the federal government?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 877,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "A federal government is the common or national government of a federation. The United States is considered the first modern federation. After declaring independence from Britain, the U.S. adopted its first constitution, the Articles of Confederation in 1781. This was the first step towards federalism by establishing the confederal Congress. However, Congress was limited as to its ability to pursue economic, military, and judiciary reform. In 1787, a Constitutional Convention drafted the United States Constitution during the Philadelphia Convention. After the ratification of the Constitution by nine states in 1788, the U.S. was officially a federation, putting the U.S. in a unique position where the central government exists by the sufferance of the individual states rather than the reverse."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        184,
        185
      ],
      "audit": {
        "id": "boolq-dev-0877",
        "label": "yes",
        "category": "ambiguous",
        "evidence": "Passage equates federal and national government within a federation. The question leaves country and constitutional structure unspecified, so universal equivalence is not established."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2630",
      "question": "are chickpeas and garbonzo beans the same thing?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2630,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The chickpea or chick pea (Cicer arietinum) is a legume of the family Fabaceae, subfamily Faboideae. Its different types are variously known as gram or Bengal gram, garbanzo or garbanzo bean, and Egyptian pea. Chickpea seeds are high in protein. It is one of the earliest cultivated legumes and 7500-year-old remains have been found in the Middle East."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        354,
        355
      ],
      "audit": {
        "id": "boolq-dev-2630",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly lists garbanzo bean as another name for chickpea."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0448",
      "question": "is every step you take about a stalker?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 448,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Sting later said he was disconcerted by how many people think the song is more positive than it is. He insists it is about the obsession with a lost lover, and the jealousy and surveillance that follow. ``One couple told me 'Oh we love that song; it was the main song played at our wedding!' I thought, 'Well, good luck.''' When asked why he appears angry in the music video, Sting told BBC Radio 2, ``I think the song is very, very sinister and ugly and people have actually misinterpreted it as being a gentle little love song, when it's quite the opposite.''"
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        68,
        69
      ],
      "audit": {
        "id": "boolq-dev-0448",
        "label": "yes",
        "category": "ambiguous",
        "evidence": "Passage describes a song about obsession, jealousy and surveillance, consistent with stalking. It never names the song, so the question's exact title/reference cannot be established from this passage alone."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-3134",
      "question": "can a person be a citizen of no country?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 3134,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "In International law a stateless person is someone who is ``not considered as a national by any state under the operation of its law''. Some stateless persons are also refugees. However, not all refugees are stateless, and many persons who are stateless have never crossed an international border."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        322,
        323
      ],
      "audit": {
        "id": "boolq-dev-3134",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage defines stateless people as not considered nationals by any state."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0525",
      "question": "are canadian dollars the same as american dollars?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 525,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The Canadian dollar (symbol: $; code: CAD; French: dollar canadien) is the currency of Canada. It is abbreviated with the dollar sign $, or sometimes Can$ or C$ to distinguish it from other dollar-denominated currencies. It is divided into 100 cents (¢)."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        162,
        163
      ],
      "audit": {
        "id": "boolq-dev-0525",
        "label": "no",
        "category": "supported",
        "evidence": "Passage identifies CAD as Canada's currency and distinguishes it from other dollar-denominated currencies."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1381",
      "question": "can a catcher block home plate without the ball?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1381,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "In baseball, blocking the plate is a technique performed by a catcher to prevent a runner from scoring. The act of blocking the plate accounted for most of the physical contact in Major League Baseball prior to the 2014 season, when it was outlawed except when the catcher already has possession of the ball."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        276,
        277
      ],
      "audit": {
        "id": "boolq-dev-1381",
        "label": "no",
        "category": "time_or_scope",
        "evidence": "Passage prohibits plate blocking without the ball in MLB from the 2014 season. The question does not specify league or date; other rulesets are not established."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0308",
      "question": "are ground coriander and cumin the same thing?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 308,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Coriander is commonly found both as whole dried seeds and in ground form. Roasting or heating the seeds in a dry pan heightens the flavour, aroma, and pungency. Ground coriander seed loses flavour quickly in storage and is best ground fresh. Coriander seed is a spice in garam masala and Indian curries which often employ the ground fruits in generous amounts together with cumin, acting as a thickener in a mixture called dhana jeera. Roasted coriander seeds, called dhana dal, are eaten as a snack. They are the main ingredient of the two south Indian dishes sambhar and rasam."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        232,
        233
      ],
      "audit": {
        "id": "boolq-dev-0308",
        "label": "no",
        "category": "supported",
        "evidence": "Passage describes coriander being used together with cumin, treating them as separate ingredients."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2849",
      "question": "is there always a stinger in a bee sting?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2849,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Although it is widely believed that a worker honey bee can sting only once, this is a partial misconception: although the stinger is in fact barbed so that it lodges in the victim's skin, tearing loose from the bee's abdomen and leading to its death in minutes, this only happens if the skin of the victim is sufficiently thick, such as a mammal's. Honey bees are the only hymenoptera with a strongly barbed sting, though yellow jackets and some other wasps have small barbs."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        282,
        283
      ],
      "audit": {
        "id": "boolq-dev-2849",
        "label": "no",
        "category": "ambiguous",
        "evidence": "Passage supports that the stinger is not always left lodged in skin. If stinger in a bee sting means the apparatus that delivers the sting, the question asks something different. The intended retained-stinger meaning is unstated."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0210",
      "question": "did murphy brown have a baby on the show?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 210,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "In the show's 1991--92 season, Murphy became pregnant. When her baby's father (ex-husband and current underground radical Jake Lowenstein) expressed his unwillingness to give up his own lifestyle to be a parent, Murphy chose to have the child and raise it alone. Another major fiction-reality blending came at Murphy's baby shower: the invited guests were journalists Katie Couric, Joan Lunden, Paula Zahn, Mary Alice Williams and Faith Daniels, who treated the fictional Murphy and Corky as friends and peers."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        280,
        281
      ],
      "audit": {
        "id": "boolq-dev-0210",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage describes pregnancy, choosing to have the child, and the baby shower; supports the intended storyline answer."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-3115",
      "question": "can i have dual citizenship in us and czech republic?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 3115,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The citizenship law of the Czech Republic is based on the principles of jus sanguinis or ``right by blood''. In other words, descent from a Czech parent is the primary method of acquiring Czech citizenship (together with naturalisation). Birth on Czech territory without a Czech parent is in itself insufficient for the conferral of Czech citizenship. Every Czech citizen is also a citizen of the European Union. The law came into effect on 1 January 1993, the date of the dissolution of Czechoslovakia, and has been amended in 1993, 1995, 1996, 1999, 2002, 2003 and 2005. Since 1 January 2014, multiple citizenship under Czech law is allowed."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        258,
        259
      ],
      "audit": {
        "id": "boolq-dev-3115",
        "label": "yes",
        "category": "ambiguous",
        "evidence": "Passage permits multiple citizenship under Czech law from 2014 but does not state US rules or the questioner's eligibility. Yes is plausible for possibility, but a personal can-I conclusion is not established by the passage."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0587",
      "question": "is boost mobile and virgin mobile the same?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 587,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "At the 2010 Consumer Electronics Show, Boost Mobile announced it would begin to offer a new unlimited plan using Sprint's CDMA network, costing $50 a month. For $10 more, Boost also offered an unlimited plan for the BlackBerry Curve 8830. Sprint would also acquire fellow prepaid wireless provider Virgin Mobile USA in 2010--both Boost and Virgin Mobile would be re-organized into a new group within Sprint, encompassing the two brands and other no-contract phone services offered by the company."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        352,
        353
      ],
      "audit": {
        "id": "boolq-dev-0587",
        "label": "yes",
        "category": "ambiguous",
        "evidence": "Passage says Boost and Virgin were two brands reorganized into one Sprint group in 2010. Same parent versus same brand/service gives different readings; the ownership frame is also dated."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2552",
      "question": "is lily-of-the-valley poisonous?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2552,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Lily of the valley (Convallaria majalis /ˌkɒnvəˈleɪriə məˈdʒeɪlɪs/), sometimes written lily-of-the-valley, is a sweetly scented, highly poisonous woodland flowering plant that is native throughout the cool temperate Northern Hemisphere in Asia and Europe. Other names include May bells, Our Lady's tears, and Mary's tears. Its French name, muguet, sometimes appears in the names of perfumes imitating the flower's scent."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        108,
        109
      ],
      "audit": {
        "id": "boolq-dev-2552",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly calls lily-of-the-valley highly poisonous."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1859",
      "question": "are pinot noir and pinot gris the same grape?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1859,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Pinot gris, pinot grigio or Grauburgunder is a white wine grape variety of the species Vitis vinifera. Thought to be a mutant clone of the pinot noir variety, it normally has a grayish-blue fruit, accounting for its name but the grapes can have a brownish pink to black and even white appearance. The word pinot could have been given to it because the grapes grow in small pine cone-shaped clusters. The wines produced from this grape also vary in color from a deep golden yellow to copper and even a light shade of pink, and it is one of the more popular grapes for skin-contact wine."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        140,
        141
      ],
      "audit": {
        "id": "boolq-dev-1859",
        "label": "no",
        "category": "supported",
        "evidence": "Passage describes Pinot gris as a distinct variety thought to be a mutant clone of Pinot noir."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1739",
      "question": "is there spongy bone in the medullary cavity?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1739,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The medullary cavity (medulla, innermost part) is the central cavity of bone shafts where red bone marrow and/or yellow bone marrow (adipose tissue) is stored; hence, the medullary cavity is also known as the marrow cavity. Located in the main shaft of a long bone (diaphysis) (consisting mostly of compact bone), the medullary cavity has walls composed of spongy bone (cancellous bone) and is lined with a thin, vascular membrane (endosteum). However, the medullary cavity is the area inside any bone (long, flat, etc.) that holds the bone marrow."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        222,
        223
      ],
      "audit": {
        "id": "boolq-dev-1739",
        "label": "yes",
        "category": "ambiguous",
        "evidence": "Passage places spongy bone in the walls of the medullary cavity and marrow in the space itself. In the cavity can include its walls or mean only its contents; yes adopts the former."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1066",
      "question": "is the preamble a part of the indian constitution?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1066,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The preamble to the Constitution of India is a brief introductory statement that sets out the guiding people and principles of the document, and it indicates the source from which the ordinary document derives its authority, meaning, the people. The hopes and aspirations of the people as well as the ideals before our nation are described in the preamble in clear words. It may be considered as the heart and soul of Constitution. The preamble can be referred to as the preface which highlights the entire Constitution. It was adopted on 26 November 1949 by the Constituent Assembly and came into effect on 26th January 1950."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        12,
        13
      ],
      "audit": {
        "id": "boolq-dev-1066",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage describes the preamble as the Constitution's introductory statement, adopted by the Constituent Assembly with it."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-3247",
      "question": "is there a season 5 of arrested development?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 3247,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The fifth season of the television comedy series Arrested Development premiered on Netflix on May 29, 2018. The season will consist of 16 episodes, split into two eight-episode parts; with the second half premiering later in 2018. This is the second revival season after the series was canceled by Fox in 2006; the fourth season premiered in 2013."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        294,
        295
      ],
      "audit": {
        "id": "boolq-dev-3247",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly says the fifth season premiered in 2018; this establishes existence without relying on its future-release details."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1335",
      "question": "are there nerves in the central nervous system?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1335,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The central nervous system (CNS) is the part of the nervous system consisting of the brain and spinal cord. The central nervous system is so named because it integrates information it receives from, and coordinates and influences the activity of, all parts of the bodies of bilaterally symmetric animals--that is, all multicellular animals except sponges and radially symmetric animals such as jellyfish--and it contains the majority of the nervous system. Many consider the retina and the optic nerve (cranial nerve II), as well as the olfactory nerves (cranial nerve I) and olfactory epithelium as parts of the CNS, synapsing directly on brain tissue without intermediate ganglia. As such, the olfactory epithelium is the only central nervous tissue in direct contact with the environment, which opens up for therapeutic treatments. The CNS is contained within the dorsal body cavity, with the brain housed in the cranial cavity and the spinal cord in the spinal canal. In vertebrates, the brain is protected by the skull, while the spinal cord is protected by the vertebrae. The brain and spinal cord are both enclosed in the meninges. In central nervous systems, the interneuronal space is filled with a large amount of supporting non-nervous cells called neuroglial cells."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        148,
        149
      ],
      "audit": {
        "id": "boolq-dev-1335",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly includes optic and olfactory nerves as parts of the CNS under a stated common classification."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1979",
      "question": "is there such a thing as insulin resistance?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1979,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Insulin resistance (IR) is a pathological condition in which cells fail to respond normally to the hormone insulin. The body produces insulin when glucose starts to be released into the bloodstream from the digestion of carbohydrates (primarily) in the diet. Under normal conditions of insulin reactivity, this insulin response triggers glucose being taken into body cells, to be used for energy, and inhibits the body from using fat for energy, thereby causing the concentration of glucose in the blood to decrease as a result, staying within the normal range even when a large amount of carbohydrates is consumed. During insulin resistance, however, excess glucose is not sufficiently absorbed by cells even in the presence of insulin, thereby causing an increase in the level of blood sugar."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        46,
        47
      ],
      "audit": {
        "id": "boolq-dev-1979",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage defines insulin resistance as a pathological condition."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2426",
      "question": "do you have the right to remain silent before arrest?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2426,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Outside the context of lawful detention or arrest, a person has no duty to answer any questions of the police. If judicial compulsion is sought by the State, the person can still invoke his or her Fifth Amendment right against compulsory self-incrimination, and refuse to testify if answers to questions posed are potentially self-incriminating. Only if granted immunity by the state, in a formal proceeding, from having any testimony or evidence derived from the testimony used against him or her, can a person be compelled to answer over an assertion of this right. If police detain (or arrest) a person, they must advise him or her that he or she has a right to remain silent, and the right to an attorney, among other rights. (This is known as the Miranda warning.) If the detained person invokes these rights, all interrogation must cease, and ordinarily nothing said by the defendant in violation of this rule may be admitted against him or her at trial."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        250,
        251
      ],
      "audit": {
        "id": "boolq-dev-2426",
        "label": "yes",
        "category": "time_or_scope",
        "evidence": "Passage explicitly states no duty to answer police questions outside detention or arrest, in a US/Fifth-Amendment context. The question does not name a jurisdiction; no current legal verification was performed."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-3165",
      "question": "is red stripe light sold in the us?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 3165,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "In the U.S. in 2010, the bottling size was reduced from a typical 12 oz. per serving to 11.2 oz. per serving which is equivalent to the typical metric serving of 0.33L."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        94,
        95
      ],
      "audit": {
        "id": "boolq-dev-3165",
        "label": "yes",
        "category": "ambiguous",
        "evidence": "Passage only describes a 2010 US bottling-size reduction. It names no product and does not distinguish Red Stripe Light from another beverage or current availability. The yes label is unsupported by this excerpt alone."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0980",
      "question": "is pyruvic acid and pyruvate the same thing?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 980,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Pyruvic acid (CHCOCOOH) is the simplest of the alpha-keto acids, with a carboxylic acid and a ketone functional group. Pyruvate (/paɪˈruːveɪt/), the conjugate base, CHCOCOO, is a key intermediate in several metabolic pathways."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        164,
        165
      ],
      "audit": {
        "id": "boolq-dev-0980",
        "label": "no",
        "category": "supported",
        "evidence": "Passage identifies pyruvate as the conjugate base of pyruvic acid, distinguishing the chemical forms."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2011",
      "question": "did dwight howard play for the houston rockets?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2011,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Dwight David Howard (born December 8, 1985) is an American professional basketball player for the Washington Wizards of the National Basketball Association (NBA). Howard, who plays center, spent his high school career at Southwest Atlanta Christian Academy. He chose to forgo college and entered the 2004 NBA draft, and was selected first overall by the Orlando Magic. An eight-time All-Star, eight-time All-NBA Team honoree, five-time All-Defensive member, and three-time Defensive Player of the Year, Howard has been ranked consistently as one of the best in the league in rebounds, blocks, and field goal percentage, and has set numerous franchise and league records. In 2009, he led the Magic to the NBA Finals. In 2012, after eight seasons with the Magic, Howard was traded to the Los Angeles Lakers. After one season with the Lakers, he joined the Houston Rockets. He spent three seasons with the Rockets, before joining the Atlanta Hawks in 2016, and being traded to the Hornets in 2017. In July 2018, he was traded once again, this time to the Brooklyn Nets."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        180,
        181
      ],
      "audit": {
        "id": "boolq-dev-2011",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage explicitly says Howard joined Houston and spent three seasons with the Rockets."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-2818",
      "question": "is there a way one can grow more crops from the same land?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 2818,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Agronomists describe the benefits to yield in rotated crops as ``The Rotation Effect''. There are many found benefits of rotation systems: however, there is no specific scientific basis for the sometimes 10-25% yield increase in a crop grown in rotation versus monoculture. The factors related to the increase are simply described as alleviation of the negative factors of monoculture cropping systems. Explanations due to improved nutrition; pest, pathogen, and weed stress reduction; and improved soil structure have been found in some cases to be correlated, but causation has not been determined for the majority of cropping systems."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        328,
        329
      ],
      "audit": {
        "id": "boolq-dev-2818",
        "label": "yes",
        "category": "supported",
        "evidence": "Passage reports yield increases under rotation; uncertain causal explanation does not negate the stated increase."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1603",
      "question": "is father's day the same day all over the world?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1603,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Father's Day is a celebration honoring fathers and celebrating fatherhood, paternal bonds, and the influence of fathers in society. In Catholic Europe, it has been celebrated on March 19 (St. Joseph's Day) since the Middle Ages. This celebration was brought by the Spanish and Portuguese to Latin America, where March 19 is often still used for it, though many countries in Europe and the Americas have adopted the U.S. date, which is the third Sunday of June. It is celebrated on various days in many parts of the world, most commonly in the months of March, April and June. It complements similar celebrations honoring family members, such as Mother's Day, Siblings Day, and Grandparents' Day."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        296,
        297
      ],
      "audit": {
        "id": "boolq-dev-1603",
        "label": "no",
        "category": "supported",
        "evidence": "Passage explicitly describes Father's Day on different dates across countries."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1625",
      "question": "is japanese black garlic supposed to be mushy?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1625,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "Black garlic can be eaten alone, on bread, or used in soups, sauces, crushed into a mayonnaise or simply tossed into a vegetable dish. A vinaigrette can be made with black garlic, sherry vinegar, soy, a neutral oil, and Dijon mustard. Its softness increases with water content."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        76,
        77
      ],
      "audit": {
        "id": "boolq-dev-1625",
        "label": "yes",
        "category": "ambiguous",
        "evidence": "Passage says black garlic softness increases with water content. It does not establish a normal or desired mushy texture, or a specifically Japanese preparation."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-0767",
      "question": "is a soccer stadium bigger than a football stadium?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 767,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "A soccer-specific stadium typically has amenities, dimensions and scale suitable for soccer in North America, including a scoreboard, video screen, luxury suites and possibly a roof. The field dimensions are within the range found optimal by FIFA: 110--120 yards (100--110 m) long by 70--80 yards (64--73 m) wide. These soccer field dimensions are wider than the regulation American football field width of 53 ⁄ yards (48.8 m), or the 65-yard (59 m) width of a Canadian football field. The playing surface typically consists of grass as opposed to artificial turf, as the latter is generally disfavored for soccer matches since players are more susceptible to injuries. However, some soccer specific stadiums, such as Portland's Providence Park and Creighton University's Morrison Stadium, do have artificial turf."
      },
      "target": 1.0,
      "target_kind": "label",
      "records": [
        28,
        29
      ],
      "audit": {
        "id": "boolq-dev-0767",
        "label": "yes",
        "category": "ambiguous",
        "evidence": "Passage compares playing-field widths, not whole-stadium size or seating capacity. Bigger stadium is broader than the evidence; yes assumes a field-size interpretation."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The answer to the question is no, using the supplied passage as evidence.",
        "true": "The answer to the question is yes, using the supplied passage as evidence."
      },
      "experiment": "boolq",
      "id": "boolq-dev-1570",
      "question": "was the iphone the first touch screen phone?",
      "source": {
        "authors": "Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova",
        "changes": "Selected subset; added trailing question mark; converted boolean label to 0/1; wrapped passage in state.",
        "dataset": "BoolQ",
        "downloaded_mirror": "https://huggingface.co/datasets/google/boolq",
        "kind": "published_human_label",
        "license": "CC BY-SA 3.0",
        "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
        "original_download": "https://storage.googleapis.com/boolq/dev.jsonl",
        "paper": "https://arxiv.org/abs/1905.10044",
        "publisher": "Google Research",
        "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
        "repository": "https://github.com/google-research-datasets/boolean-questions",
        "row_index_zero_based": 1570,
        "rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
        "split": "development"
      },
      "state": {
        "passage": "The first commercially available device that could be properly referred to as a ``smartphone'' began as a prototype called ``Angler'' developed by Frank Canova in 1992 while at IBM and demonstrated in November of that year at the COMDEX computer industry trade show. A refined version was marketed to consumers in 1994 by BellSouth under the name Simon Personal Communicator. In addition to placing and receiving cellular calls, the touchscreen-equipped Simon could send and receive faxes and emails. It included an address book, calendar, appointment scheduler, calculator, world time clock, and notepad, as well as other visionary mobile applications such as maps, stock reports and news. The term ``smart phone'' or ``smartphone'' was not coined until a year after the introduction of the Simon, appearing in print as early as 1995, describing AT&T's PhoneWriter Communicator."
      },
      "target": 0.0,
      "target_kind": "label",
      "records": [
        42,
        43
      ],
      "audit": {
        "id": "boolq-dev-1570",
        "label": "no",
        "category": "ambiguous",
        "evidence": "Passage establishes a touchscreen phone in 1994 but gives no iPhone release date. The no answer requires background knowledge not contained in the supplied excerpt."
      },
      "family": "boolq"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-001",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 2,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 11,
            "unmarked": 30
          },
          "B": {
            "marked": 15,
            "unmarked": 15
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.5,
      "target_kind": "probability",
      "records": [
        198,
        199
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-002",
      "question": "Are both selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 117,
        "truth_numerator": 92,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 24,
          "unmarked": 3
        },
        "event": "both_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.7863247863247863,
      "target_kind": "probability",
      "records": [
        268,
        269
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-003",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 1,
        "truth_numerator": 0,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 22,
            "unmarked": 53
          },
          "B": {
            "marked": 0,
            "unmarked": 60
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.0,
      "target_kind": "probability",
      "records": [
        338,
        339
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-004",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 5,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 7,
            "unmarked": 9
          },
          "B": {
            "marked": 4,
            "unmarked": 16
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.2,
      "target_kind": "probability",
      "records": [
        288,
        289
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-005",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 40,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 1,
            "unmarked": 39
          },
          "B": {
            "marked": 7,
            "unmarked": 26
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "A",
        "selection": "Choose one item uniformly at random from group A only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.025,
      "target_kind": "probability",
      "records": [
        130,
        131
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-006",
      "question": "Is exactly one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 25,
        "truth_numerator": 7,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 4,
          "unmarked": 21
        },
        "event": "exactly_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.28,
      "target_kind": "probability",
      "records": [
        284,
        285
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-007",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 368,
        "truth_numerator": 267,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 3,
          "B": 9
        },
        "boxes": {
          "A": {
            "marked": 7,
            "unmarked": 39
          },
          "B": {
            "marked": 33,
            "unmarked": 3
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.7255434782608695,
      "target_kind": "probability",
      "records": [
        22,
        23
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-008",
      "question": "Is at least one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 65,
        "truth_numerator": 63,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 21,
          "unmarked": 5
        },
        "event": "at_least_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.9692307692307692,
      "target_kind": "probability",
      "records": [
        344,
        345
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-009",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 1,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 11,
            "unmarked": 69
          },
          "B": {
            "marked": 30,
            "unmarked": 0
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 1.0,
      "target_kind": "probability",
      "records": [
        230,
        231
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-010",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 2,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 8,
            "unmarked": 29
          },
          "B": {
            "marked": 20,
            "unmarked": 20
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.5,
      "target_kind": "probability",
      "records": [
        182,
        183
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-011",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 940,
        "truth_numerator": 509,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 9,
          "B": 6
        },
        "boxes": {
          "A": {
            "marked": 6,
            "unmarked": 18
          },
          "B": {
            "marked": 46,
            "unmarked": 1
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.5414893617021277,
      "target_kind": "probability",
      "records": [
        24,
        25
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-012",
      "question": "Is at least one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 703,
        "truth_numerator": 625,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 25,
          "unmarked": 13
        },
        "event": "at_least_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.8890469416785206,
      "target_kind": "probability",
      "records": [
        316,
        317
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-013",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 20,
        "truth_numerator": 19,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 51,
            "unmarked": 11
          },
          "B": {
            "marked": 19,
            "unmarked": 1
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.95,
      "target_kind": "probability",
      "records": [
        320,
        321
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-014",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 20,
        "truth_numerator": 19,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 37,
            "unmarked": 14
          },
          "B": {
            "marked": 19,
            "unmarked": 1
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.95,
      "target_kind": "probability",
      "records": [
        84,
        85
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-015",
      "question": "Is at least one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 155,
        "truth_numerator": 38,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 4,
          "unmarked": 27
        },
        "event": "at_least_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.24516129032258063,
      "target_kind": "probability",
      "records": [
        172,
        173
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-016",
      "question": "Are both selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 253,
        "truth_numerator": 6,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 4,
          "unmarked": 19
        },
        "event": "both_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.023715415019762844,
      "target_kind": "probability",
      "records": [
        160,
        161
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-017",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 60,
        "truth_numerator": 31,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 7,
            "unmarked": 10
          },
          "B": {
            "marked": 31,
            "unmarked": 29
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.5166666666666667,
      "target_kind": "probability",
      "records": [
        106,
        107
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-018",
      "question": "Is at least one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 41,
        "truth_numerator": 30,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 20,
          "unmarked": 22
        },
        "event": "at_least_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.7317073170731707,
      "target_kind": "probability",
      "records": [
        48,
        49
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-019",
      "question": "Is at least one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 15,
        "truth_numerator": 14,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 7,
          "unmarked": 3
        },
        "event": "at_least_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.9333333333333333,
      "target_kind": "probability",
      "records": [
        356,
        357
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-020",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 207,
        "truth_numerator": 113,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 1,
          "B": 2
        },
        "boxes": {
          "A": {
            "marked": 14,
            "unmarked": 32
          },
          "B": {
            "marked": 16,
            "unmarked": 8
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.5458937198067633,
      "target_kind": "probability",
      "records": [
        90,
        91
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-021",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 5,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 13,
            "unmarked": 64
          },
          "B": {
            "marked": 10,
            "unmarked": 40
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.2,
      "target_kind": "probability",
      "records": [
        132,
        133
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-022",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 936,
        "truth_numerator": 653,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 3,
          "B": 3
        },
        "boxes": {
          "A": {
            "marked": 17,
            "unmarked": 19
          },
          "B": {
            "marked": 12,
            "unmarked": 1
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.6976495726495726,
      "target_kind": "probability",
      "records": [
        138,
        139
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-023",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 91,
        "truth_numerator": 18,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 4,
          "B": 9
        },
        "boxes": {
          "A": {
            "marked": 18,
            "unmarked": 10
          },
          "B": {
            "marked": 0,
            "unmarked": 5
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.1978021978021978,
      "target_kind": "probability",
      "records": [
        136,
        137
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-024",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 1225,
        "truth_numerator": 828,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 3,
          "B": 2
        },
        "boxes": {
          "A": {
            "marked": 4,
            "unmarked": 1
          },
          "B": {
            "marked": 24,
            "unmarked": 25
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.6759183673469388,
      "target_kind": "probability",
      "records": [
        96,
        97
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-025",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 1,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 50,
            "unmarked": 0
          },
          "B": {
            "marked": 20,
            "unmarked": 5
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "A",
        "selection": "Choose one item uniformly at random from group A only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 1.0,
      "target_kind": "probability",
      "records": [
        26,
        27
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-026",
      "question": "Is at least one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 8,
        "truth_numerator": 7,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 10,
          "unmarked": 6
        },
        "event": "at_least_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.875,
      "target_kind": "probability",
      "records": [
        228,
        229
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-027",
      "question": "Is exactly one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 30,
        "truth_numerator": 7,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 2,
          "unmarked": 14
        },
        "event": "exactly_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.23333333333333334,
      "target_kind": "probability",
      "records": [
        142,
        143
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-028",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 4902,
        "truth_numerator": 4049,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 1,
          "B": 5
        },
        "boxes": {
          "A": {
            "marked": 8,
            "unmarked": 11
          },
          "B": {
            "marked": 39,
            "unmarked": 4
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.8259893920848633,
      "target_kind": "probability",
      "records": [
        100,
        101
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-029",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 208,
        "truth_numerator": 185,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 7,
          "B": 6
        },
        "boxes": {
          "A": {
            "marked": 45,
            "unmarked": 3
          },
          "B": {
            "marked": 30,
            "unmarked": 6
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.8894230769230769,
      "target_kind": "probability",
      "records": [
        272,
        273
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-030",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 24,
        "truth_numerator": 11,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 3,
          "B": 1
        },
        "boxes": {
          "A": {
            "marked": 6,
            "unmarked": 12
          },
          "B": {
            "marked": 5,
            "unmarked": 1
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.4583333333333333,
      "target_kind": "probability",
      "records": [
        20,
        21
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-031",
      "question": "Is exactly one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 28,
        "truth_numerator": 15,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 3,
          "unmarked": 5
        },
        "event": "exactly_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.5357142857142857,
      "target_kind": "probability",
      "records": [
        36,
        37
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-032",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 5,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 4,
            "unmarked": 16
          },
          "B": {
            "marked": 11,
            "unmarked": 2
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "A",
        "selection": "Choose one item uniformly at random from group A only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.2,
      "target_kind": "probability",
      "records": [
        114,
        115
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-033",
      "question": "Is exactly one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 95,
        "truth_numerator": 42,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 6,
          "unmarked": 14
        },
        "event": "exactly_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.4421052631578947,
      "target_kind": "probability",
      "records": [
        38,
        39
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-034",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 50,
        "truth_numerator": 49,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 49,
            "unmarked": 1
          },
          "B": {
            "marked": 9,
            "unmarked": 19
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "A",
        "selection": "Choose one item uniformly at random from group A only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.98,
      "target_kind": "probability",
      "records": [
        214,
        215
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-035",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 5,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 4,
            "unmarked": 16
          },
          "B": {
            "marked": 32,
            "unmarked": 1
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "A",
        "selection": "Choose one item uniformly at random from group A only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.2,
      "target_kind": "probability",
      "records": [
        244,
        245
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-036",
      "question": "Are both selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 105,
        "truth_numerator": 11,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 12,
          "unmarked": 24
        },
        "event": "both_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.10476190476190476,
      "target_kind": "probability",
      "records": [
        336,
        337
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-037",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 30,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 21,
            "unmarked": 27
          },
          "B": {
            "marked": 1,
            "unmarked": 29
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.03333333333333333,
      "target_kind": "probability",
      "records": [
        334,
        335
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-038",
      "question": "Are both selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 1,
        "truth_numerator": 0,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 1,
          "unmarked": 5
        },
        "event": "both_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.0,
      "target_kind": "probability",
      "records": [
        156,
        157
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-039",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 5,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 0,
            "unmarked": 16
          },
          "B": {
            "marked": 12,
            "unmarked": 48
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.2,
      "target_kind": "probability",
      "records": [
        54,
        55
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-040",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 9860,
        "truth_numerator": 8201,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 1,
          "B": 9
        },
        "boxes": {
          "A": {
            "marked": 19,
            "unmarked": 15
          },
          "B": {
            "marked": 25,
            "unmarked": 4
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.8317444219066937,
      "target_kind": "probability",
      "records": [
        92,
        93
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-041",
      "question": "Is exactly one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 1,
        "truth_numerator": 0,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 0,
          "unmarked": 21
        },
        "event": "exactly_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.0,
      "target_kind": "probability",
      "records": [
        252,
        253
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-042",
      "question": "Is exactly one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 1,
        "truth_numerator": 0,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 0,
          "unmarked": 18
        },
        "event": "exactly_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.0,
      "target_kind": "probability",
      "records": [
        70,
        71
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-043",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 27,
        "truth_numerator": 25,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 4,
          "B": 5
        },
        "boxes": {
          "A": {
            "marked": 5,
            "unmarked": 0
          },
          "B": {
            "marked": 39,
            "unmarked": 6
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.9259259259259259,
      "target_kind": "probability",
      "records": [
        206,
        207
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-044",
      "question": "Are both selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 15,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 2,
          "unmarked": 4
        },
        "event": "both_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.06666666666666667,
      "target_kind": "probability",
      "records": [
        6,
        7
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-045",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 2,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 20,
            "unmarked": 20
          },
          "B": {
            "marked": 2,
            "unmarked": 54
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "A",
        "selection": "Choose one item uniformly at random from group A only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.5,
      "target_kind": "probability",
      "records": [
        194,
        195
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-046",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 2109,
        "truth_numerator": 298,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 2,
          "B": 4
        },
        "boxes": {
          "A": {
            "marked": 4,
            "unmarked": 33
          },
          "B": {
            "marked": 3,
            "unmarked": 16
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.14129919393077287,
      "target_kind": "probability",
      "records": [
        196,
        197
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-047",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 48,
        "truth_numerator": 37,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 3,
          "B": 6
        },
        "boxes": {
          "A": {
            "marked": 5,
            "unmarked": 11
          },
          "B": {
            "marked": 27,
            "unmarked": 0
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.7708333333333334,
      "target_kind": "probability",
      "records": [
        346,
        347
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-048",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 90,
        "truth_numerator": 7,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 8,
          "B": 7
        },
        "boxes": {
          "A": {
            "marked": 7,
            "unmarked": 41
          },
          "B": {
            "marked": 0,
            "unmarked": 8
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.07777777777777778,
      "target_kind": "probability",
      "records": [
        122,
        123
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-049",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 6235,
        "truth_numerator": 3599,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 7,
          "B": 3
        },
        "boxes": {
          "A": {
            "marked": 19,
            "unmarked": 10
          },
          "B": {
            "marked": 17,
            "unmarked": 26
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.5772253408179631,
      "target_kind": "probability",
      "records": [
        248,
        249
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-050",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 60,
        "truth_numerator": 31,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 33,
            "unmarked": 8
          },
          "B": {
            "marked": 31,
            "unmarked": 29
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "B",
        "selection": "Choose one item uniformly at random from group B only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.5166666666666667,
      "target_kind": "probability",
      "records": [
        234,
        235
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-051",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 2,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 30,
            "unmarked": 30
          },
          "B": {
            "marked": 5,
            "unmarked": 38
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "A",
        "selection": "Choose one item uniformly at random from group A only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.5,
      "target_kind": "probability",
      "records": [
        200,
        201
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-052",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 312,
        "truth_numerator": 151,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 7,
          "B": 9
        },
        "boxes": {
          "A": {
            "marked": 3,
            "unmarked": 15
          },
          "B": {
            "marked": 19,
            "unmarked": 7
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.483974358974359,
      "target_kind": "probability",
      "records": [
        110,
        111
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-053",
      "question": "Is exactly one of the two selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 561,
        "truth_numerator": 280,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 14,
          "unmarked": 20
        },
        "event": "exactly_one_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.49910873440285203,
      "target_kind": "probability",
      "records": [
        218,
        219
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-054",
      "question": "Are both selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 42,
        "truth_numerator": 1,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 6,
          "unmarked": 30
        },
        "event": "both_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.023809523809523808,
      "target_kind": "probability",
      "records": [
        360,
        361
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-055",
      "question": "Are both selected items marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "without_replacement",
        "seed": 20261002,
        "truth_denominator": 78,
        "truth_numerator": 55,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "marked": 11,
          "unmarked": 2
        },
        "event": "both_marked",
        "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
        "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
        "setting": "A fictional box contains the entire population described by these counts."
      },
      "target": 0.7051282051282052,
      "target_kind": "probability",
      "records": [
        56,
        57
      ],
      "audit": null,
      "family": "without_replacement"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-056",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 540,
        "truth_numerator": 407,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 8,
          "B": 4
        },
        "boxes": {
          "A": {
            "marked": 21,
            "unmarked": 9
          },
          "B": {
            "marked": 31,
            "unmarked": 5
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.7537037037037037,
      "target_kind": "probability",
      "records": [
        112,
        113
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-057",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "conditional_table",
        "seed": 20261002,
        "truth_denominator": 20,
        "truth_numerator": 11,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "counts": {
          "A": {
            "marked": 11,
            "unmarked": 9
          },
          "B": {
            "marked": 5,
            "unmarked": 5
          }
        },
        "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
        "selected_group": "A",
        "selection": "Choose one item uniformly at random from group A only.",
        "setting": "Fictional inventory. The following table describes the entire population, not a sample."
      },
      "target": 0.55,
      "target_kind": "probability",
      "records": [
        30,
        31
      ],
      "audit": null,
      "family": "conditional_table"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-058",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 819,
        "truth_numerator": 548,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 8,
          "B": 1
        },
        "boxes": {
          "A": {
            "marked": 30,
            "unmarked": 12
          },
          "B": {
            "marked": 4,
            "unmarked": 9
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.6691086691086691,
      "target_kind": "probability",
      "records": [
        298,
        299
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-059",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 1595,
        "truth_numerator": 359,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 2,
          "B": 3
        },
        "boxes": {
          "A": {
            "marked": 1,
            "unmarked": 21
          },
          "B": {
            "marked": 10,
            "unmarked": 19
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.22507836990595612,
      "target_kind": "probability",
      "records": [
        220,
        221
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "The stated event does not occur in that random selection.",
        "true": "The stated event occurs in the random selection described in the state."
      },
      "experiment": "probability",
      "id": "probability-060",
      "question": "Is the selected item marked?",
      "source": {
        "generator": "datasets.py",
        "generator_version": 1,
        "kind": "synthetic_exact_probability",
        "probability_family": "weighted_mixture",
        "seed": 20261002,
        "truth_denominator": 1519,
        "truth_numerator": 753,
        "validation": "Exact rational probability from the complete population and stated random selection mechanism"
      },
      "state": {
        "box_selection_tickets": {
          "A": 6,
          "B": 3
        },
        "boxes": {
          "A": {
            "marked": 23,
            "unmarked": 26
          },
          "B": {
            "marked": 17,
            "unmarked": 14
          }
        },
        "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
        "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
        "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
      },
      "target": 0.4957208689927584,
      "target_kind": "probability",
      "records": [
        170,
        171
      ],
      "audit": null,
      "family": "weighted_mixture"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-induction-boundary",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "optical",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 365,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 8,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 180,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        389,
        390,
        391
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-induction-expired",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "optical",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 366,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": true,
          "emergency": true,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        380,
        381,
        382
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-trainee-supervised",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "portable",
          "category_endorsed": false,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 4,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "trainee",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": true,
          "supervisor_category_endorsed": true,
          "supervisor_present": true,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        413,
        414,
        415
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-trainee-remote",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "portable",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "trainee",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": true,
          "supervisor_category_endorsed": true,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        374,
        375,
        376
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-amber-exception",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "thermal",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": true,
          "emergency": true,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 0,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "amber",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        428,
        429,
        430
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-red-no-exception",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "thermal",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": true,
          "emergency": true,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "red",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        434,
        435,
        436
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-after-hours-permitted",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": true,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "rotating",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 2,
          "duty_manager_signed": true,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 2,
          "pickup_staffed": false,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        365,
        366,
        367
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-after-hours-unsigned",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": true,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "rotating",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": false,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        401,
        402,
        403
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-sensor-outdoor-boundary",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 20,
          "calibration_current": true,
          "category": "sensor",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": false,
          "duration_hours": 72,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": -20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 3000,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        368,
        369,
        370
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-sensor-outdoor-weather",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "sensor",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": false,
          "duration_hours": 1,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": false
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        398,
        399,
        400
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-portable-weather-irrelevant",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 30,
          "calibration_current": false,
          "category": "portable",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": false,
          "duration_hours": 24,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": false,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": false
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        371,
        372,
        373
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-portable-battery-low",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 29.9,
          "calibration_current": true,
          "category": "portable",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": true,
          "emergency": true,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        422,
        423,
        424
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-sterile-limits",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 10,
          "calibration_current": true,
          "category": "sterile",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 6,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 25,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 300,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        419,
        420,
        421
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-sterile-broken-seal",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "sterile",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": true,
          "emergency": true,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": false,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        392,
        393,
        394
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-member-supervised-expired",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "sensor",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 5,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 181,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": true,
          "supervisor_category_endorsed": true,
          "supervisor_present": true,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        383,
        384,
        385
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-trainee-duration-limit",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "sensor",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 4.1,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "trainee",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": true,
          "supervisor_category_endorsed": true,
          "supervisor_present": true,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        410,
        411,
        412
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-optical-no-transport",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "optical",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": false,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 10,
          "fire_watch": false,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": false,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 0,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        377,
        378,
        379
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-optical-dirty-route",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "optical",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": false,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 1,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        425,
        426,
        427
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-maintenance-last-day",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 2,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "thermal",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 35,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 1,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        386,
        387,
        388
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-maintenance-due",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "thermal",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": true,
          "emergency": true,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 0,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        416,
        417,
        418
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-sensor-indoor-weather",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 2,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 20,
          "calibration_current": true,
          "category": "sensor",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": false
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        395,
        396,
        397
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-reservation-no-waiver",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "sensor",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": true,
          "emergency": true,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 1,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        431,
        432,
        433
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-rotating-limits",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 0,
          "calibration_current": true,
          "category": "rotating",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 2,
          "duty_manager_signed": false,
          "emergency": false,
          "endorsement_age_days": 30,
          "environment_temperature_c": 5,
          "fire_watch": true,
          "guard_fitted": true,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 100,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 1,
      "target_kind": "label",
      "records": [
        404,
        405,
        406
      ],
      "audit": null,
      "family": "cache_policy"
    },
    {
      "criteria": {
        "false": "At least one applicable requirement fails.",
        "true": "Every applicable requirement passes, including any precisely stated exceptions."
      },
      "experiment": "cache_policy",
      "id": "cache-rotating-no-guard",
      "question": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
      "state": {
        "facts": {
          "active_other_loans": 0,
          "after_hours_permit": false,
          "area_exclusion": true,
          "badge_active": true,
          "battery_percent": 100,
          "calibration_current": true,
          "category": "rotating",
          "category_endorsed": true,
          "clean_destination": true,
          "clean_route": true,
          "days_since_induction": 30,
          "destination_approved": true,
          "destination_indoor": true,
          "duration_hours": 1,
          "duty_manager_signed": true,
          "emergency": true,
          "endorsement_age_days": 30,
          "environment_temperature_c": 20,
          "fire_watch": true,
          "guard_fitted": false,
          "heat_shield": true,
          "known_damage": false,
          "maintenance_days_remaining": 10,
          "membership_active": true,
          "notice_hours": 4,
          "pickup_staffed": true,
          "power_supply_tested": true,
          "protective_case": true,
          "release_tag": "green",
          "requester_role": "member",
          "reservation_conflicts": 0,
          "seal_intact": true,
          "sterile_witness": true,
          "supervisor_active": false,
          "supervisor_category_endorsed": false,
          "supervisor_present": false,
          "transport_distance_m": 50,
          "unresolved_suspension": false,
          "vibration_isolation": true,
          "weather_clear": true
        },
        "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n"
      },
      "target": 0,
      "target_kind": "label",
      "records": [
        407,
        408,
        409
      ],
      "audit": null,
      "family": "cache_policy"
    }
  ],
  "records": [
    {
      "case_id": "warmup-yes",
      "experiment": "warmup",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "warmup",
      "sequence": 0,
      "started_at": "2026-10-02T08:42:23.245615+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The rule does not require approval.\", \"true\": \"The rule requires approval.\"}, \"instructions\": \"Is approval required under the rule?\", \"type\": \"noul\"}}, \"state\": {\"number\": 15, \"rule\": \"Approve if the number is at least 10.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3e-05,
      "usage": {
        "prompt_tokens": 200,
        "completion_tokens": 20,
        "total_tokens": 220,
        "cost": 3e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3e-05,
          "upstream_inference_prompt_cost": 2e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930543-yjGh71kw3YzClXOxTGg9",
        "object": "chat.completion",
        "created": 1790930543,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 200,
          "completion_tokens": 20,
          "total_tokens": 220,
          "cost": 3e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3e-05,
            "upstream_inference_prompt_cost": 2e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930543-yjGh71kw3YzClXOxTGg9",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1338991670054384,
      "run": "main",
      "record_index": 0
    },
    {
      "case_id": "warmup-yes",
      "experiment": "warmup",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "warmup",
      "sequence": 1,
      "started_at": "2026-10-02T08:42:24.380359+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rule": "Approve if the number is at least 10.",
          "number": 15
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is approval required under the rule?",
            "criteria": {
              "true": "The rule requires approval.",
              "false": "The rule does not require approval."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.96,
      "valid": true,
      "error": null,
      "cost_usd": 1.3692e-05,
      "usage": {
        "input_tokens": 326,
        "output_tokens": 20,
        "cost": 1.3692e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.96
          }
        },
        "usage": {
          "input_tokens": 326,
          "output_tokens": 20,
          "cost": 1.3692e-05
        },
        "id": "gen-dec-1790930544-OGjFOl5MpdR9bNc9p4kP",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930544-OGjFOl5MpdR9bNc9p4kP",
      "answer": {
        "type": "noul",
        "noul": 0.96
      },
      "latency_s": 0.24059216599562205,
      "run": "main",
      "record_index": 1
    },
    {
      "case_id": "warmup-no",
      "experiment": "warmup",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "warmup",
      "sequence": 2,
      "started_at": "2026-10-02T08:42:24.621444+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rule": "Approve if the number is at least 10.",
          "number": 4
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is approval required under the rule?",
            "criteria": {
              "true": "The rule requires approval.",
              "false": "The rule does not require approval."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.04,
      "valid": true,
      "error": null,
      "cost_usd": 1.365e-05,
      "usage": {
        "input_tokens": 325,
        "output_tokens": 20,
        "cost": 1.365e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.04
          }
        },
        "usage": {
          "input_tokens": 325,
          "output_tokens": 20,
          "cost": 1.365e-05
        },
        "id": "gen-dec-1790930544-wseDwJoFGP2zCpXh5cDW",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930544-wseDwJoFGP2zCpXh5cDW",
      "answer": {
        "type": "noul",
        "noul": 0.04
      },
      "latency_s": 0.251195041986648,
      "run": "main",
      "record_index": 2
    },
    {
      "case_id": "warmup-no",
      "experiment": "warmup",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "warmup",
      "sequence": 3,
      "started_at": "2026-10-02T08:42:24.872968+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The rule does not require approval.\", \"true\": \"The rule requires approval.\"}, \"instructions\": \"Is approval required under the rule?\", \"type\": \"noul\"}}, \"state\": {\"number\": 4, \"rule\": \"Approve if the number is at least 10.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3e-05,
      "usage": {
        "prompt_tokens": 200,
        "completion_tokens": 20,
        "total_tokens": 220,
        "cost": 3e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3e-05,
          "upstream_inference_prompt_cost": 2e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930544-fSyzxPgaWKJaDtBuFrc8",
        "object": "chat.completion",
        "created": 1790930544,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 200,
          "completion_tokens": 20,
          "total_tokens": 220,
          "cost": 3e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3e-05,
            "upstream_inference_prompt_cost": 2e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930544-fSyzxPgaWKJaDtBuFrc8",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.97343324997928,
      "run": "main",
      "record_index": 3
    },
    {
      "case_id": "policy-016",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 4,
      "started_at": "2026-10-02T08:42:26.846915+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 610, \"active_promotion\": false, \"gift_wrapping\": true, \"hazardous\": true, \"mainland\": true, \"member\": false, \"subtotal_credits\": 99, \"weight_kg\": 20}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.64e-05,
      "usage": {
        "prompt_tokens": 364,
        "completion_tokens": 20,
        "total_tokens": 384,
        "cost": 4.64e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.64e-05,
          "upstream_inference_prompt_cost": 3.64e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930546-U6lnBJYVKYWiukoOMAV1",
        "object": "chat.completion",
        "created": 1790930546,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 364,
          "completion_tokens": 20,
          "total_tokens": 384,
          "cost": 4.64e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.64e-05,
            "upstream_inference_prompt_cost": 3.64e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930546-U6lnBJYVKYWiukoOMAV1",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.8997180419974029,
      "run": "main",
      "record_index": 4
    },
    {
      "case_id": "policy-016",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 5,
      "started_at": "2026-10-02T08:42:27.747158+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 610,
            "active_promotion": false,
            "gift_wrapping": true,
            "hazardous": true,
            "mainland": true,
            "member": false,
            "subtotal_credits": 99,
            "weight_kg": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.06,
      "valid": true,
      "error": null,
      "cost_usd": 2.1252e-05,
      "usage": {
        "input_tokens": 506,
        "output_tokens": 20,
        "cost": 2.1252e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.06
          }
        },
        "usage": {
          "input_tokens": 506,
          "output_tokens": 20,
          "cost": 2.1252e-05
        },
        "id": "gen-dec-1790930547-9rj2QEbDNR0OSZNGeH0e",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930547-9rj2QEbDNR0OSZNGeH0e",
      "answer": {
        "type": "noul",
        "noul": 0.06
      },
      "latency_s": 0.23809362499741837,
      "run": "main",
      "record_index": 5
    },
    {
      "case_id": "probability-044",
      "experiment": "probability",
      "target": 0.06666666666666667,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 6,
      "started_at": "2026-10-02T08:42:27.985629+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 2,
            "unmarked": 4
          },
          "event": "both_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Are both selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.19,
      "valid": true,
      "error": null,
      "cost_usd": 1.7262e-05,
      "usage": {
        "input_tokens": 411,
        "output_tokens": 20,
        "cost": 1.7262e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.19
          }
        },
        "usage": {
          "input_tokens": 411,
          "output_tokens": 20,
          "cost": 1.7262e-05
        },
        "id": "gen-dec-1790930548-Cbk4M7i6VCWQlz2WyWDn",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930548-Cbk4M7i6VCWQlz2WyWDn",
      "answer": {
        "type": "noul",
        "noul": 0.19
      },
      "latency_s": 0.21925979098887183,
      "run": "main",
      "record_index": 6
    },
    {
      "case_id": "probability-044",
      "experiment": "probability",
      "target": 0.06666666666666667,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 7,
      "started_at": "2026-10-02T08:42:28.205350+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Are both selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 2, \"unmarked\": 4}, \"event\": \"both_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.06666666666666667,
      "valid": true,
      "error": null,
      "cost_usd": 4.09e-05,
      "usage": {
        "prompt_tokens": 274,
        "completion_tokens": 27,
        "total_tokens": 301,
        "cost": 4.09e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.09e-05,
          "upstream_inference_prompt_cost": 2.74e-05,
          "upstream_inference_completions_cost": 1.35e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930548-YVtYHcGZTpf8reLwrmAe",
        "object": "chat.completion",
        "created": 1790930548,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.06666666666666667}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 274,
          "completion_tokens": 27,
          "total_tokens": 301,
          "cost": 4.09e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.09e-05,
            "upstream_inference_prompt_cost": 2.74e-05,
            "upstream_inference_completions_cost": 1.35e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930548-YVtYHcGZTpf8reLwrmAe",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.06666666666666667
      },
      "latency_s": 1.0398890420037787,
      "run": "main",
      "record_index": 7
    },
    {
      "case_id": "boolq-dev-1550",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 8,
      "started_at": "2026-10-02T08:42:29.245743+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is australia part of the ring of fire?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The Ring of Fire is a direct result of plate tectonics: the movement and collisions of lithospheric plates. The eastern section of the ring is the result of the Nazca Plate and the Cocos Plate being subducted beneath the westward-moving South American Plate. The Cocos Plate is being subducted beneath the Caribbean Plate, in Central America. A portion of the Pacific Plate and the small Juan de Fuca Plate are being subducted beneath the North American Plate. Along the northern portion, the northwestward-moving Pacific plate is being subducted beneath the Aleutian Islands arc. Farther west, the Pacific plate is being subducted along the Kamchatka Peninsula arcs on south past Japan. The southern portion is more complex, with a number of smaller tectonic plates in collision with the Pacific plate from the Mariana Islands, the Philippines, Bougainville, Tonga, and New Zealand; this portion excludes Australia, since it lies in the center of its tectonic plate. Indonesia lies between the Ring of Fire along the northeastern islands adjacent to and including New Guinea and the Alpide belt along the south and west from Sumatra, Java, Bali, Flores, and Timor.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 5.48e-05,
      "usage": {
        "prompt_tokens": 448,
        "completion_tokens": 20,
        "total_tokens": 468,
        "cost": 5.48e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.48e-05,
          "upstream_inference_prompt_cost": 4.48e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930549-Lls1mLtMY17BYli1ZFPQ",
        "object": "chat.completion",
        "created": 1790930549,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 448,
          "completion_tokens": 20,
          "total_tokens": 468,
          "cost": 5.48e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 5.48e-05,
            "upstream_inference_prompt_cost": 4.48e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930549-Lls1mLtMY17BYli1ZFPQ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.3291317500115838,
      "run": "main",
      "record_index": 8
    },
    {
      "case_id": "boolq-dev-1550",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 9,
      "started_at": "2026-10-02T08:42:30.575544+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The Ring of Fire is a direct result of plate tectonics: the movement and collisions of lithospheric plates. The eastern section of the ring is the result of the Nazca Plate and the Cocos Plate being subducted beneath the westward-moving South American Plate. The Cocos Plate is being subducted beneath the Caribbean Plate, in Central America. A portion of the Pacific Plate and the small Juan de Fuca Plate are being subducted beneath the North American Plate. Along the northern portion, the northwestward-moving Pacific plate is being subducted beneath the Aleutian Islands arc. Farther west, the Pacific plate is being subducted along the Kamchatka Peninsula arcs on south past Japan. The southern portion is more complex, with a number of smaller tectonic plates in collision with the Pacific plate from the Mariana Islands, the Philippines, Bougainville, Tonga, and New Zealand; this portion excludes Australia, since it lies in the center of its tectonic plate. Indonesia lies between the Ring of Fire along the northeastern islands adjacent to and including New Guinea and the Alpide belt along the south and west from Sumatra, Java, Bali, Flores, and Timor."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is australia part of the ring of fire?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.06,
      "valid": true,
      "error": null,
      "cost_usd": 2.4192e-05,
      "usage": {
        "input_tokens": 576,
        "output_tokens": 20,
        "cost": 2.4192e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.06
          }
        },
        "usage": {
          "input_tokens": 576,
          "output_tokens": 20,
          "cost": 2.4192e-05
        },
        "id": "gen-dec-1790930550-JBukCRVjG0ebpZGCtSLS",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930550-JBukCRVjG0ebpZGCtSLS",
      "answer": {
        "type": "noul",
        "noul": 0.06
      },
      "latency_s": 0.34738187497714534,
      "run": "main",
      "record_index": 9
    },
    {
      "case_id": "boolq-dev-0579",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 10,
      "started_at": "2026-10-02T08:42:30.923291+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Its plumage is lavender-blue to mid-blue in the crest, back, wings, and tail, and its face is white. The underside is off-white and the neck is collared with black which extends to the sides of the head. The wing primaries and tail are strongly barred with black, sky-blue and white. The bill, legs, and eyes are all black. Males and females are almost identical, but the male is slightly larger."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "do female and male blue jays look the same?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.63,
      "valid": true,
      "error": null,
      "cost_usd": 1.7724e-05,
      "usage": {
        "input_tokens": 422,
        "output_tokens": 20,
        "cost": 1.7724e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.63
          }
        },
        "usage": {
          "input_tokens": 422,
          "output_tokens": 20,
          "cost": 1.7724e-05
        },
        "id": "gen-dec-1790930550-YRepaabsUa8dKnewPJwo",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930550-YRepaabsUa8dKnewPJwo",
      "answer": {
        "type": "noul",
        "noul": 0.63
      },
      "latency_s": 0.24034704201039858,
      "run": "main",
      "record_index": 10
    },
    {
      "case_id": "boolq-dev-0579",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 11,
      "started_at": "2026-10-02T08:42:31.163983+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"do female and male blue jays look the same?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Its plumage is lavender-blue to mid-blue in the crest, back, wings, and tail, and its face is white. The underside is off-white and the neck is collared with black which extends to the sides of the head. The wing primaries and tail are strongly barred with black, sky-blue and white. The bill, legs, and eyes are all black. Males and females are almost identical, but the male is slightly larger.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.98e-05,
      "usage": {
        "prompt_tokens": 298,
        "completion_tokens": 20,
        "total_tokens": 318,
        "cost": 3.98e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.98e-05,
          "upstream_inference_prompt_cost": 2.98e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930551-dyjlwHYgXEwKrUYTcJkM",
        "object": "chat.completion",
        "created": 1790930551,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 298,
          "completion_tokens": 20,
          "total_tokens": 318,
          "cost": 3.98e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.98e-05,
            "upstream_inference_prompt_cost": 2.98e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930551-dyjlwHYgXEwKrUYTcJkM",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.3163369170215447,
      "run": "main",
      "record_index": 11
    },
    {
      "case_id": "boolq-dev-1066",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 12,
      "started_at": "2026-10-02T08:42:32.480689+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The preamble to the Constitution of India is a brief introductory statement that sets out the guiding people and principles of the document, and it indicates the source from which the ordinary document derives its authority, meaning, the people. The hopes and aspirations of the people as well as the ideals before our nation are described in the preamble in clear words. It may be considered as the heart and soul of Constitution. The preamble can be referred to as the preface which highlights the entire Constitution. It was adopted on 26 November 1949 by the Constituent Assembly and came into effect on 26th January 1950."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is the preamble a part of the indian constitution?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.86,
      "valid": true,
      "error": null,
      "cost_usd": 1.9614e-05,
      "usage": {
        "input_tokens": 467,
        "output_tokens": 20,
        "cost": 1.9614e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.86
          }
        },
        "usage": {
          "input_tokens": 467,
          "output_tokens": 20,
          "cost": 1.9614e-05
        },
        "id": "gen-dec-1790930552-H9mbRmDkP02kvjOZXPGS",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930552-H9mbRmDkP02kvjOZXPGS",
      "answer": {
        "type": "noul",
        "noul": 0.86
      },
      "latency_s": 0.23135162502876483,
      "run": "main",
      "record_index": 12
    },
    {
      "case_id": "boolq-dev-1066",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 13,
      "started_at": "2026-10-02T08:42:32.712380+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is the preamble a part of the indian constitution?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The preamble to the Constitution of India is a brief introductory statement that sets out the guiding people and principles of the document, and it indicates the source from which the ordinary document derives its authority, meaning, the people. The hopes and aspirations of the people as well as the ideals before our nation are described in the preamble in clear words. It may be considered as the heart and soul of Constitution. The preamble can be referred to as the preface which highlights the entire Constitution. It was adopted on 26 November 1949 by the Constituent Assembly and came into effect on 26th January 1950.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 4.43e-05,
      "usage": {
        "prompt_tokens": 333,
        "completion_tokens": 22,
        "total_tokens": 355,
        "cost": 4.43e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.43e-05,
          "upstream_inference_prompt_cost": 3.33e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930552-miZPmrDYHUAVM1znuERa",
        "object": "chat.completion",
        "created": 1790930552,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.99}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 333,
          "completion_tokens": 22,
          "total_tokens": 355,
          "cost": 4.43e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.43e-05,
            "upstream_inference_prompt_cost": 3.33e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930552-miZPmrDYHUAVM1znuERa",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.8990407919918653,
      "run": "main",
      "record_index": 13
    },
    {
      "case_id": "policy-021",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 14,
      "started_at": "2026-10-02T08:42:33.611877+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": false,
            "badge_color": "blue",
            "clearance": 3,
            "emergency_permit": false,
            "experience_years": 19,
            "role": "visitor",
            "sponsor": false,
            "suspended": true,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.04,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.04
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930553-YzdebeYmFowG5a1fdTq2",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930553-YzdebeYmFowG5a1fdTq2",
      "answer": {
        "type": "noul",
        "noul": 0.04
      },
      "latency_s": 0.22194620899972506,
      "run": "main",
      "record_index": 14
    },
    {
      "case_id": "policy-021",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 15,
      "started_at": "2026-10-02T08:42:33.834178+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": false, \"badge_color\": \"blue\", \"clearance\": 3, \"emergency_permit\": false, \"experience_years\": 19, \"role\": \"visitor\", \"sponsor\": false, \"suspended\": true, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930553-JxWMMnT5I4mKv3x09CEG",
        "object": "chat.completion",
        "created": 1790930553,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930553-JxWMMnT5I4mKv3x09CEG",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9450915409834124,
      "run": "main",
      "record_index": 15
    },
    {
      "case_id": "boolq-dev-1114",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 16,
      "started_at": "2026-10-02T08:42:34.779711+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The gray wolf (Canis lupus), also known as the timber wolf, western wolf, or simply, wolf, is a canine native to the wilderness and remote areas of Eurasia and North America. It is the largest extant member of its family, with males averaging 43--45 kg (95--99 lb) and females 36--38.5 kg (79--85 lb). Like the red wolf, it is distinguished from other Canis species by its larger size and less pointed features, particularly on the ears and muzzle. Its winter fur is long and bushy and predominantly a mottled gray in color, although nearly pure white, red, and brown to black also occur. Mammal Species of the World (3rd ed., 2005), a standard reference work in zoology, recognises 38 subspecies of C. lupus.."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is a timber wolf the same as a gray wolf?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 2.1924e-05,
      "usage": {
        "input_tokens": 522,
        "output_tokens": 20,
        "cost": 2.1924e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 522,
          "output_tokens": 20,
          "cost": 2.1924e-05
        },
        "id": "gen-dec-1790930554-oaaxDHodcEXFaNqhFqCG",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930554-oaaxDHodcEXFaNqhFqCG",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.22352170801605098,
      "run": "main",
      "record_index": 16
    },
    {
      "case_id": "boolq-dev-1114",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 17,
      "started_at": "2026-10-02T08:42:35.003646+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is a timber wolf the same as a gray wolf?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The gray wolf (Canis lupus), also known as the timber wolf, western wolf, or simply, wolf, is a canine native to the wilderness and remote areas of Eurasia and North America. It is the largest extant member of its family, with males averaging 43--45 kg (95--99 lb) and females 36--38.5 kg (79--85 lb). Like the red wolf, it is distinguished from other Canis species by its larger size and less pointed features, particularly on the ears and muzzle. Its winter fur is long and bushy and predominantly a mottled gray in color, although nearly pure white, red, and brown to black also occur. Mammal Species of the World (3rd ed., 2005), a standard reference work in zoology, recognises 38 subspecies of C. lupus..\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.81e-05,
      "usage": {
        "prompt_tokens": 381,
        "completion_tokens": 20,
        "total_tokens": 401,
        "cost": 4.81e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.81e-05,
          "upstream_inference_prompt_cost": 3.81e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930555-7UH0Z1Ejw2JE9yTqj18R",
        "object": "chat.completion",
        "created": 1790930555,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 381,
          "completion_tokens": 20,
          "total_tokens": 401,
          "cost": 4.81e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.81e-05,
            "upstream_inference_prompt_cost": 3.81e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930555-7UH0Z1Ejw2JE9yTqj18R",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0752395410090685,
      "run": "main",
      "record_index": 17
    },
    {
      "case_id": "boolq-dev-3114",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 18,
      "started_at": "2026-10-02T08:42:36.079308+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is a midsummer night's dream a comedy?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"A Midsummer Night's Dream is a comedy written by William Shakespeare in 1595/96. It portrays the events surrounding the marriage of Theseus, the Duke of Athens, to Hippolyta, the former queen of the Amazons. These include the adventures of four young Athenian lovers and a group of six amateur actors (the mechanicals) who are controlled and manipulated by the fairies who inhabit the forest in which most of the play is set. The play is one of Shakespeare's most popular works for the stage and is widely performed across the world.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.21e-05,
      "usage": {
        "prompt_tokens": 321,
        "completion_tokens": 20,
        "total_tokens": 341,
        "cost": 4.21e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.21e-05,
          "upstream_inference_prompt_cost": 3.21e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930556-xo4t7zTY09YqbROwgx2z",
        "object": "chat.completion",
        "created": 1790930556,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 321,
          "completion_tokens": 20,
          "total_tokens": 341,
          "cost": 4.21e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.21e-05,
            "upstream_inference_prompt_cost": 3.21e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930556-xo4t7zTY09YqbROwgx2z",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.4670071250002366,
      "run": "main",
      "record_index": 18
    },
    {
      "case_id": "boolq-dev-3114",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 19,
      "started_at": "2026-10-02T08:42:37.546806+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "A Midsummer Night's Dream is a comedy written by William Shakespeare in 1595/96. It portrays the events surrounding the marriage of Theseus, the Duke of Athens, to Hippolyta, the former queen of the Amazons. These include the adventures of four young Athenian lovers and a group of six amateur actors (the mechanicals) who are controlled and manipulated by the fairies who inhabit the forest in which most of the play is set. The play is one of Shakespeare's most popular works for the stage and is widely performed across the world."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is a midsummer night's dream a comedy?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 1.9068e-05,
      "usage": {
        "input_tokens": 454,
        "output_tokens": 20,
        "cost": 1.9068e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 454,
          "output_tokens": 20,
          "cost": 1.9068e-05
        },
        "id": "gen-dec-1790930557-y78UMMWv88U65xaroCtw",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930557-y78UMMWv88U65xaroCtw",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.24637766598607413,
      "run": "main",
      "record_index": 19
    },
    {
      "case_id": "probability-030",
      "experiment": "probability",
      "target": 0.4583333333333333,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 20,
      "started_at": "2026-10-02T08:42:37.793514+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 3,
            "B": 1
          },
          "boxes": {
            "A": {
              "marked": 6,
              "unmarked": 12
            },
            "B": {
              "marked": 5,
              "unmarked": 1
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.41,
      "valid": true,
      "error": null,
      "cost_usd": 1.995e-05,
      "usage": {
        "input_tokens": 475,
        "output_tokens": 20,
        "cost": 1.995e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.41
          }
        },
        "usage": {
          "input_tokens": 475,
          "output_tokens": 20,
          "cost": 1.995e-05
        },
        "id": "gen-dec-1790930557-bxo4cstWExHf33IMvOeV",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930557-bxo4cstWExHf33IMvOeV",
      "answer": {
        "type": "noul",
        "noul": 0.41
      },
      "latency_s": 0.21927012500236742,
      "run": "main",
      "record_index": 20
    },
    {
      "case_id": "probability-030",
      "experiment": "probability",
      "target": 0.4583333333333333,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 21,
      "started_at": "2026-10-02T08:42:38.013232+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 3, \"B\": 1}, \"boxes\": {\"A\": {\"marked\": 6, \"unmarked\": 12}, \"B\": {\"marked\": 5, \"unmarked\": 1}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.4166666667,
      "valid": true,
      "error": null,
      "cost_usd": 4.54e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 25,
        "total_tokens": 354,
        "cost": 4.54e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.54e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930558-lqkDQHpsgDULm6ajiOv0",
        "object": "chat.completion",
        "created": 1790930558,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.4166666667}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 25,
          "total_tokens": 354,
          "cost": 4.54e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.54e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930558-lqkDQHpsgDULm6ajiOv0",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.4166666667
      },
      "latency_s": 1.0648542499984615,
      "run": "main",
      "record_index": 21
    },
    {
      "case_id": "probability-007",
      "experiment": "probability",
      "target": 0.7255434782608695,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 22,
      "started_at": "2026-10-02T08:42:39.078512+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 3, \"B\": 9}, \"boxes\": {\"A\": {\"marked\": 7, \"unmarked\": 39}, \"B\": {\"marked\": 33, \"unmarked\": 3}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.8214285714,
      "valid": true,
      "error": null,
      "cost_usd": 4.54e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 25,
        "total_tokens": 354,
        "cost": 4.54e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.54e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930559-dnckx1BpZH5jvkv3yT8k",
        "object": "chat.completion",
        "created": 1790930559,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.8214285714}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 25,
          "total_tokens": 354,
          "cost": 4.54e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.54e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930559-dnckx1BpZH5jvkv3yT8k",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.8214285714
      },
      "latency_s": 1.2597768749983516,
      "run": "main",
      "record_index": 22
    },
    {
      "case_id": "probability-007",
      "experiment": "probability",
      "target": 0.7255434782608695,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 23,
      "started_at": "2026-10-02T08:42:40.338864+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 3,
            "B": 9
          },
          "boxes": {
            "A": {
              "marked": 7,
              "unmarked": 39
            },
            "B": {
              "marked": 33,
              "unmarked": 3
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.49,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.49
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930560-9SA06oLqZA5NjGLDGF91",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930560-9SA06oLqZA5NjGLDGF91",
      "answer": {
        "type": "noul",
        "noul": 0.49
      },
      "latency_s": 0.24390487498021685,
      "run": "main",
      "record_index": 23
    },
    {
      "case_id": "probability-011",
      "experiment": "probability",
      "target": 0.5414893617021277,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 24,
      "started_at": "2026-10-02T08:42:40.583182+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 9,
            "B": 6
          },
          "boxes": {
            "A": {
              "marked": 6,
              "unmarked": 18
            },
            "B": {
              "marked": 46,
              "unmarked": 1
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.59,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.59
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930560-9r714qUCbph9eM7QJw6S",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930560-9r714qUCbph9eM7QJw6S",
      "answer": {
        "type": "noul",
        "noul": 0.59
      },
      "latency_s": 0.2270914169785101,
      "run": "main",
      "record_index": 24
    },
    {
      "case_id": "probability-011",
      "experiment": "probability",
      "target": 0.5414893617021277,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 25,
      "started_at": "2026-10-02T08:42:40.810758+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 9, \"B\": 6}, \"boxes\": {\"A\": {\"marked\": 6, \"unmarked\": 18}, \"B\": {\"marked\": 46, \"unmarked\": 1}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.584,
      "valid": true,
      "error": null,
      "cost_usd": 4.39e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 22,
        "total_tokens": 351,
        "cost": 4.39e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.39e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930560-EEjYYAP6KcMhjhmOa5J6",
        "object": "chat.completion",
        "created": 1790930560,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.584}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 22,
          "total_tokens": 351,
          "cost": 4.39e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.39e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930560-EEjYYAP6KcMhjhmOa5J6",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.584
      },
      "latency_s": 1.412256499985233,
      "run": "main",
      "record_index": 25
    },
    {
      "case_id": "probability-025",
      "experiment": "probability",
      "target": 1.0,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 26,
      "started_at": "2026-10-02T08:42:42.223457+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 50, \"unmarked\": 0}, \"B\": {\"marked\": 20, \"unmarked\": 5}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"A\", \"selection\": \"Choose one item uniformly at random from group A only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.91e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 20,
        "total_tokens": 311,
        "cost": 3.91e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.91e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930562-WH0uEPMzAPdFrTIJWfcD",
        "object": "chat.completion",
        "created": 1790930562,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 20,
          "total_tokens": 311,
          "cost": 3.91e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.91e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930562-WH0uEPMzAPdFrTIJWfcD",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.060862249985803,
      "run": "main",
      "record_index": 26
    },
    {
      "case_id": "probability-025",
      "experiment": "probability",
      "target": 1.0,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 27,
      "started_at": "2026-10-02T08:42:43.284917+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 50,
              "unmarked": 0
            },
            "B": {
              "marked": 20,
              "unmarked": 5
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "A",
          "selection": "Choose one item uniformly at random from group A only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 1.8396e-05,
      "usage": {
        "input_tokens": 438,
        "output_tokens": 20,
        "cost": 1.8396e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 438,
          "output_tokens": 20,
          "cost": 1.8396e-05
        },
        "id": "gen-dec-1790930563-x3DZXCGzxx3XSgQXCi6z",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930563-x3DZXCGzxx3XSgQXCi6z",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.22329904200159945,
      "run": "main",
      "record_index": 27
    },
    {
      "case_id": "boolq-dev-0767",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 28,
      "started_at": "2026-10-02T08:42:43.508758+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is a soccer stadium bigger than a football stadium?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"A soccer-specific stadium typically has amenities, dimensions and scale suitable for soccer in North America, including a scoreboard, video screen, luxury suites and possibly a roof. The field dimensions are within the range found optimal by FIFA: 110--120 yards (100--110 m) long by 70--80 yards (64--73 m) wide. These soccer field dimensions are wider than the regulation American football field width of 53 ⁄ yards (48.8 m), or the 65-yard (59 m) width of a Canadian football field. The playing surface typically consists of grass as opposed to artificial turf, as the latter is generally disfavored for soccer matches since players are more susceptible to injuries. However, some soccer specific stadiums, such as Portland's Providence Park and Creighton University's Morrison Stadium, do have artificial turf.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5,
      "valid": true,
      "error": null,
      "cost_usd": 4.85e-05,
      "usage": {
        "prompt_tokens": 375,
        "completion_tokens": 22,
        "total_tokens": 397,
        "cost": 4.85e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.85e-05,
          "upstream_inference_prompt_cost": 3.75e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930563-5WadQ7KD8W6Na0PxFNXW",
        "object": "chat.completion",
        "created": 1790930563,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 375,
          "completion_tokens": 22,
          "total_tokens": 397,
          "cost": 4.85e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.85e-05,
            "upstream_inference_prompt_cost": 3.75e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930563-5WadQ7KD8W6Na0PxFNXW",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5
      },
      "latency_s": 0.9867166659969371,
      "run": "main",
      "record_index": 28
    },
    {
      "case_id": "boolq-dev-0767",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 29,
      "started_at": "2026-10-02T08:42:44.495961+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "A soccer-specific stadium typically has amenities, dimensions and scale suitable for soccer in North America, including a scoreboard, video screen, luxury suites and possibly a roof. The field dimensions are within the range found optimal by FIFA: 110--120 yards (100--110 m) long by 70--80 yards (64--73 m) wide. These soccer field dimensions are wider than the regulation American football field width of 53 ⁄ yards (48.8 m), or the 65-yard (59 m) width of a Canadian football field. The playing surface typically consists of grass as opposed to artificial turf, as the latter is generally disfavored for soccer matches since players are more susceptible to injuries. However, some soccer specific stadiums, such as Portland's Providence Park and Creighton University's Morrison Stadium, do have artificial turf."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is a soccer stadium bigger than a football stadium?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.62,
      "valid": true,
      "error": null,
      "cost_usd": 2.1672e-05,
      "usage": {
        "input_tokens": 516,
        "output_tokens": 20,
        "cost": 2.1672e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.62
          }
        },
        "usage": {
          "input_tokens": 516,
          "output_tokens": 20,
          "cost": 2.1672e-05
        },
        "id": "gen-dec-1790930564-7tT64XQjkScJ3zLq2xRC",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930564-7tT64XQjkScJ3zLq2xRC",
      "answer": {
        "type": "noul",
        "noul": 0.62
      },
      "latency_s": 0.26086787501117215,
      "run": "main",
      "record_index": 29
    },
    {
      "case_id": "probability-057",
      "experiment": "probability",
      "target": 0.55,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 30,
      "started_at": "2026-10-02T08:42:44.757413+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 11, \"unmarked\": 9}, \"B\": {\"marked\": 5, \"unmarked\": 5}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"A\", \"selection\": \"Choose one item uniformly at random from group A only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.55,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930564-YE6pIEIUUZuhOoJegIv8",
        "object": "chat.completion",
        "created": 1790930564,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.55}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930564-YE6pIEIUUZuhOoJegIv8",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.55
      },
      "latency_s": 0.9881264160212595,
      "run": "main",
      "record_index": 30
    },
    {
      "case_id": "probability-057",
      "experiment": "probability",
      "target": 0.55,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 31,
      "started_at": "2026-10-02T08:42:45.746077+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 11,
              "unmarked": 9
            },
            "B": {
              "marked": 5,
              "unmarked": 5
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "A",
          "selection": "Choose one item uniformly at random from group A only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.56,
      "valid": true,
      "error": null,
      "cost_usd": 1.8354e-05,
      "usage": {
        "input_tokens": 437,
        "output_tokens": 20,
        "cost": 1.8354e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.56
          }
        },
        "usage": {
          "input_tokens": 437,
          "output_tokens": 20,
          "cost": 1.8354e-05
        },
        "id": "gen-dec-1790930565-HpArBj1q8sKEZ7yFQvNw",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930565-HpArBj1q8sKEZ7yFQvNw",
      "answer": {
        "type": "noul",
        "noul": 0.56
      },
      "latency_s": 0.22516649999306537,
      "run": "main",
      "record_index": 31
    },
    {
      "case_id": "boolq-dev-1447",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 32,
      "started_at": "2026-10-02T08:42:45.971669+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "In optics, a virtual image is an image formed when the outgoing rays from a point on an object always diverge. The image appears to be located at the point of apparent divergence. Because the rays never really converge, a virtual image cannot be projected onto a screen. In diagrams of optical systems, virtual rays are conventionally represented by dotted lines. Virtual images are located by tracing the real rays that emerge from an optical device (lens, mirror, or some combination) backward to a perceived point of origin."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "can an image be formed without a screen?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.97,
      "valid": true,
      "error": null,
      "cost_usd": 1.8186e-05,
      "usage": {
        "input_tokens": 433,
        "output_tokens": 20,
        "cost": 1.8186e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.97
          }
        },
        "usage": {
          "input_tokens": 433,
          "output_tokens": 20,
          "cost": 1.8186e-05
        },
        "id": "gen-dec-1790930565-YAOt2s5UrzAvB2BttZYG",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930565-YAOt2s5UrzAvB2BttZYG",
      "answer": {
        "type": "noul",
        "noul": 0.97
      },
      "latency_s": 0.22614770798827522,
      "run": "main",
      "record_index": 32
    },
    {
      "case_id": "boolq-dev-1447",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 33,
      "started_at": "2026-10-02T08:42:46.198230+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"can an image be formed without a screen?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"In optics, a virtual image is an image formed when the outgoing rays from a point on an object always diverge. The image appears to be located at the point of apparent divergence. Because the rays never really converge, a virtual image cannot be projected onto a screen. In diagrams of optical systems, virtual rays are conventionally represented by dotted lines. Virtual images are located by tracing the real rays that emerge from an optical device (lens, mirror, or some combination) backward to a perceived point of origin.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.07e-05,
      "usage": {
        "prompt_tokens": 307,
        "completion_tokens": 20,
        "total_tokens": 327,
        "cost": 4.07e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.07e-05,
          "upstream_inference_prompt_cost": 3.07e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930566-9HTHVRsjyBVjUQa0TiY7",
        "object": "chat.completion",
        "created": 1790930566,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 307,
          "completion_tokens": 20,
          "total_tokens": 327,
          "cost": 4.07e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.07e-05,
            "upstream_inference_prompt_cost": 3.07e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930566-9HTHVRsjyBVjUQa0TiY7",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2599990830058232,
      "run": "main",
      "record_index": 33
    },
    {
      "case_id": "boolq-dev-2791",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 34,
      "started_at": "2026-10-02T08:42:47.458738+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is new york city and manhattan the same thing?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"New York City is split up into five boroughs, which are the Bronx, Brooklyn, Manhattan, Queens, and Staten Island. Each borough has the same boundaries as a county of the state. The county governments were dissolved when the city consolidated in 1898, along with all city, town, and village governments within each county. The term borough was adopted to describe a unique form of governmental administration for each of the five fundamental constituent parts of the newly consolidated city.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 301,
        "completion_tokens": 20,
        "total_tokens": 321,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 3.01e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930567-S3RPDUlKYc2Ukg7pTO9i",
        "object": "chat.completion",
        "created": 1790930567,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 301,
          "completion_tokens": 20,
          "total_tokens": 321,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 3.01e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930567-S3RPDUlKYc2Ukg7pTO9i",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9077726250106934,
      "run": "main",
      "record_index": 34
    },
    {
      "case_id": "boolq-dev-2791",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 35,
      "started_at": "2026-10-02T08:42:48.368889+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "New York City is split up into five boroughs, which are the Bronx, Brooklyn, Manhattan, Queens, and Staten Island. Each borough has the same boundaries as a county of the state. The county governments were dissolved when the city consolidated in 1898, along with all city, town, and village governments within each county. The term borough was adopted to describe a unique form of governmental administration for each of the five fundamental constituent parts of the newly consolidated city."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is new york city and manhattan the same thing?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.05,
      "valid": true,
      "error": null,
      "cost_usd": 1.7976e-05,
      "usage": {
        "input_tokens": 428,
        "output_tokens": 20,
        "cost": 1.7976e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.05
          }
        },
        "usage": {
          "input_tokens": 428,
          "output_tokens": 20,
          "cost": 1.7976e-05
        },
        "id": "gen-dec-1790930568-T7qDYWu4gNG03JIqefl2",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930568-T7qDYWu4gNG03JIqefl2",
      "answer": {
        "type": "noul",
        "noul": 0.05
      },
      "latency_s": 0.23036637500626966,
      "run": "main",
      "record_index": 35
    },
    {
      "case_id": "probability-031",
      "experiment": "probability",
      "target": 0.5357142857142857,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 36,
      "started_at": "2026-10-02T08:42:48.599813+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is exactly one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 3, \"unmarked\": 5}, \"event\": \"exactly_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5357142857,
      "valid": true,
      "error": null,
      "cost_usd": 4.05e-05,
      "usage": {
        "prompt_tokens": 280,
        "completion_tokens": 25,
        "total_tokens": 305,
        "cost": 4.05e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.05e-05,
          "upstream_inference_prompt_cost": 2.8e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930568-pmPrcFZEA0Rb7HD7wQ3w",
        "object": "chat.completion",
        "created": 1790930568,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5357142857}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 280,
          "completion_tokens": 25,
          "total_tokens": 305,
          "cost": 4.05e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.05e-05,
            "upstream_inference_prompt_cost": 2.8e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930568-pmPrcFZEA0Rb7HD7wQ3w",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5357142857
      },
      "latency_s": 1.0789300840115175,
      "run": "main",
      "record_index": 36
    },
    {
      "case_id": "probability-031",
      "experiment": "probability",
      "target": 0.5357142857142857,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 37,
      "started_at": "2026-10-02T08:42:49.679557+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 3,
            "unmarked": 5
          },
          "event": "exactly_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is exactly one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.52,
      "valid": true,
      "error": null,
      "cost_usd": 1.7556e-05,
      "usage": {
        "input_tokens": 418,
        "output_tokens": 20,
        "cost": 1.7556e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.52
          }
        },
        "usage": {
          "input_tokens": 418,
          "output_tokens": 20,
          "cost": 1.7556e-05
        },
        "id": "gen-dec-1790930569-Jy07g0GG6acE1CGcxzmj",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930569-Jy07g0GG6acE1CGcxzmj",
      "answer": {
        "type": "noul",
        "noul": 0.52
      },
      "latency_s": 0.34608783401199616,
      "run": "main",
      "record_index": 37
    },
    {
      "case_id": "probability-033",
      "experiment": "probability",
      "target": 0.4421052631578947,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 38,
      "started_at": "2026-10-02T08:42:50.026851+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 6,
            "unmarked": 14
          },
          "event": "exactly_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is exactly one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.56,
      "valid": true,
      "error": null,
      "cost_usd": 1.7598e-05,
      "usage": {
        "input_tokens": 419,
        "output_tokens": 20,
        "cost": 1.7598e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.56
          }
        },
        "usage": {
          "input_tokens": 419,
          "output_tokens": 20,
          "cost": 1.7598e-05
        },
        "id": "gen-dec-1790930570-LkYaZMcWvAgxA4GiKhDc",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930570-LkYaZMcWvAgxA4GiKhDc",
      "answer": {
        "type": "noul",
        "noul": 0.56
      },
      "latency_s": 0.2560630000079982,
      "run": "main",
      "record_index": 38
    },
    {
      "case_id": "probability-033",
      "experiment": "probability",
      "target": 0.4421052631578947,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 39,
      "started_at": "2026-10-02T08:42:50.283375+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is exactly one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 6, \"unmarked\": 14}, \"event\": \"exactly_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.4421052632,
      "valid": true,
      "error": null,
      "cost_usd": 4.05e-05,
      "usage": {
        "prompt_tokens": 280,
        "completion_tokens": 25,
        "total_tokens": 305,
        "cost": 4.05e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.05e-05,
          "upstream_inference_prompt_cost": 2.8e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930570-zmMnIaW7jcG27eEYEcqy",
        "object": "chat.completion",
        "created": 1790930570,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.4421052632}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 280,
          "completion_tokens": 25,
          "total_tokens": 305,
          "cost": 4.05e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.05e-05,
            "upstream_inference_prompt_cost": 2.8e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930570-zmMnIaW7jcG27eEYEcqy",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.4421052632
      },
      "latency_s": 1.0053678329859395,
      "run": "main",
      "record_index": 39
    },
    {
      "case_id": "boolq-dev-2493",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 40,
      "started_at": "2026-10-02T08:42:51.289201+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Despite the practice's name, in most countries, the arresting person is usually designated as a person with arrest powers, who need not be a citizen of the country in which they are acting. For example, in the British jurisdiction of England and Wales, the power comes from section 24A(2) of the Police and Criminal Evidence Act 1984, called ``any person arrest''. This legislation states ``any person'' has these powers, and does not state that they need to be a British citizen."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "can you make a citizen's arrest in the uk?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.96,
      "valid": true,
      "error": null,
      "cost_usd": 1.8354e-05,
      "usage": {
        "input_tokens": 437,
        "output_tokens": 20,
        "cost": 1.8354e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.96
          }
        },
        "usage": {
          "input_tokens": 437,
          "output_tokens": 20,
          "cost": 1.8354e-05
        },
        "id": "gen-dec-1790930571-8v8mN45hrFxvylCge9OX",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930571-8v8mN45hrFxvylCge9OX",
      "answer": {
        "type": "noul",
        "noul": 0.96
      },
      "latency_s": 0.2682778329763096,
      "run": "main",
      "record_index": 40
    },
    {
      "case_id": "boolq-dev-2493",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 41,
      "started_at": "2026-10-02T08:42:51.558146+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"can you make a citizen's arrest in the uk?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Despite the practice's name, in most countries, the arresting person is usually designated as a person with arrest powers, who need not be a citizen of the country in which they are acting. For example, in the British jurisdiction of England and Wales, the power comes from section 24A(2) of the Police and Criminal Evidence Act 1984, called ``any person arrest''. This legislation states ``any person'' has these powers, and does not state that they need to be a British citizen.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.1e-05,
      "usage": {
        "prompt_tokens": 310,
        "completion_tokens": 20,
        "total_tokens": 330,
        "cost": 4.1e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.1e-05,
          "upstream_inference_prompt_cost": 3.1e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930571-Dsrxj5ehk3uf8IQHqVSj",
        "object": "chat.completion",
        "created": 1790930571,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 310,
          "completion_tokens": 20,
          "total_tokens": 330,
          "cost": 4.1e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.1e-05,
            "upstream_inference_prompt_cost": 3.1e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930571-Dsrxj5ehk3uf8IQHqVSj",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1796839580056258,
      "run": "main",
      "record_index": 41
    },
    {
      "case_id": "boolq-dev-1570",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 42,
      "started_at": "2026-10-02T08:42:52.739084+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The first commercially available device that could be properly referred to as a ``smartphone'' began as a prototype called ``Angler'' developed by Frank Canova in 1992 while at IBM and demonstrated in November of that year at the COMDEX computer industry trade show. A refined version was marketed to consumers in 1994 by BellSouth under the name Simon Personal Communicator. In addition to placing and receiving cellular calls, the touchscreen-equipped Simon could send and receive faxes and emails. It included an address book, calendar, appointment scheduler, calculator, world time clock, and notepad, as well as other visionary mobile applications such as maps, stock reports and news. The term ``smart phone'' or ``smartphone'' was not coined until a year after the introduction of the Simon, appearing in print as early as 1995, describing AT&T's PhoneWriter Communicator."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "was the iphone the first touch screen phone?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.03,
      "valid": true,
      "error": null,
      "cost_usd": 2.1924e-05,
      "usage": {
        "input_tokens": 522,
        "output_tokens": 20,
        "cost": 2.1924e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.03
          }
        },
        "usage": {
          "input_tokens": 522,
          "output_tokens": 20,
          "cost": 2.1924e-05
        },
        "id": "gen-dec-1790930572-n2AAowUdSwNyiRuuNpRN",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930572-n2AAowUdSwNyiRuuNpRN",
      "answer": {
        "type": "noul",
        "noul": 0.03
      },
      "latency_s": 0.2364539580012206,
      "run": "main",
      "record_index": 42
    },
    {
      "case_id": "boolq-dev-1570",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 43,
      "started_at": "2026-10-02T08:42:52.976010+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"was the iphone the first touch screen phone?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The first commercially available device that could be properly referred to as a ``smartphone'' began as a prototype called ``Angler'' developed by Frank Canova in 1992 while at IBM and demonstrated in November of that year at the COMDEX computer industry trade show. A refined version was marketed to consumers in 1994 by BellSouth under the name Simon Personal Communicator. In addition to placing and receiving cellular calls, the touchscreen-equipped Simon could send and receive faxes and emails. It included an address book, calendar, appointment scheduler, calculator, world time clock, and notepad, as well as other visionary mobile applications such as maps, stock reports and news. The term ``smart phone'' or ``smartphone'' was not coined until a year after the introduction of the Simon, appearing in print as early as 1995, describing AT&T's PhoneWriter Communicator.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.83e-05,
      "usage": {
        "prompt_tokens": 383,
        "completion_tokens": 20,
        "total_tokens": 403,
        "cost": 4.83e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.83e-05,
          "upstream_inference_prompt_cost": 3.83e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930572-3DucnyXujbawhGjodyQx",
        "object": "chat.completion",
        "created": 1790930572,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 383,
          "completion_tokens": 20,
          "total_tokens": 403,
          "cost": 4.83e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.83e-05,
            "upstream_inference_prompt_cost": 3.83e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930572-3DucnyXujbawhGjodyQx",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0268083330010995,
      "run": "main",
      "record_index": 43
    },
    {
      "case_id": "policy-007",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 44,
      "started_at": "2026-10-02T08:42:54.003389+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"green\", \"clearance\": 2, \"emergency_permit\": false, \"experience_years\": 27, \"role\": \"contractor\", \"sponsor\": true, \"suspended\": false, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.62e-05,
      "usage": {
        "prompt_tokens": 362,
        "completion_tokens": 20,
        "total_tokens": 382,
        "cost": 4.62e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.62e-05,
          "upstream_inference_prompt_cost": 3.62e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930574-m7uORI6iLREj0Gv3xoLq",
        "object": "chat.completion",
        "created": 1790930574,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 362,
          "completion_tokens": 20,
          "total_tokens": 382,
          "cost": 4.62e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.62e-05,
            "upstream_inference_prompt_cost": 3.62e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930574-m7uORI6iLREj0Gv3xoLq",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.1425084169895854,
      "run": "main",
      "record_index": 44
    },
    {
      "case_id": "policy-007",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 45,
      "started_at": "2026-10-02T08:42:55.146409+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "green",
            "clearance": 2,
            "emergency_permit": false,
            "experience_years": 27,
            "role": "contractor",
            "sponsor": true,
            "suspended": false,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.23,
      "valid": true,
      "error": null,
      "cost_usd": 2.1126e-05,
      "usage": {
        "input_tokens": 503,
        "output_tokens": 20,
        "cost": 2.1126e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.23
          }
        },
        "usage": {
          "input_tokens": 503,
          "output_tokens": 20,
          "cost": 2.1126e-05
        },
        "id": "gen-dec-1790930575-L6tXHML3IHwZmQpsjprz",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930575-L6tXHML3IHwZmQpsjprz",
      "answer": {
        "type": "noul",
        "noul": 0.23
      },
      "latency_s": 0.23067729201284237,
      "run": "main",
      "record_index": 45
    },
    {
      "case_id": "boolq-dev-1979",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 46,
      "started_at": "2026-10-02T08:42:55.377628+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Insulin resistance (IR) is a pathological condition in which cells fail to respond normally to the hormone insulin. The body produces insulin when glucose starts to be released into the bloodstream from the digestion of carbohydrates (primarily) in the diet. Under normal conditions of insulin reactivity, this insulin response triggers glucose being taken into body cells, to be used for energy, and inhibits the body from using fat for energy, thereby causing the concentration of glucose in the blood to decrease as a result, staying within the normal range even when a large amount of carbohydrates is consumed. During insulin resistance, however, excess glucose is not sufficiently absorbed by cells even in the presence of insulin, thereby causing an increase in the level of blood sugar."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is there such a thing as insulin resistance?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 2.0034e-05,
      "usage": {
        "input_tokens": 477,
        "output_tokens": 20,
        "cost": 2.0034e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 477,
          "output_tokens": 20,
          "cost": 2.0034e-05
        },
        "id": "gen-dec-1790930575-dHejsnOsWLRLUxjZuNfK",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930575-dHejsnOsWLRLUxjZuNfK",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.220665500004543,
      "run": "main",
      "record_index": 46
    },
    {
      "case_id": "boolq-dev-1979",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 47,
      "started_at": "2026-10-02T08:42:55.598719+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is there such a thing as insulin resistance?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Insulin resistance (IR) is a pathological condition in which cells fail to respond normally to the hormone insulin. The body produces insulin when glucose starts to be released into the bloodstream from the digestion of carbohydrates (primarily) in the diet. Under normal conditions of insulin reactivity, this insulin response triggers glucose being taken into body cells, to be used for energy, and inhibits the body from using fat for energy, thereby causing the concentration of glucose in the blood to decrease as a result, staying within the normal range even when a large amount of carbohydrates is consumed. During insulin resistance, however, excess glucose is not sufficiently absorbed by cells even in the presence of insulin, thereby causing an increase in the level of blood sugar.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.5e-05,
      "usage": {
        "prompt_tokens": 350,
        "completion_tokens": 20,
        "total_tokens": 370,
        "cost": 4.5e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.5e-05,
          "upstream_inference_prompt_cost": 3.5e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930575-dRIB09lePh3H9JrNFG9D",
        "object": "chat.completion",
        "created": 1790930575,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 350,
          "completion_tokens": 20,
          "total_tokens": 370,
          "cost": 4.5e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.5e-05,
            "upstream_inference_prompt_cost": 3.5e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930575-dRIB09lePh3H9JrNFG9D",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.3523345839930698,
      "run": "main",
      "record_index": 47
    },
    {
      "case_id": "probability-018",
      "experiment": "probability",
      "target": 0.7317073170731707,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 48,
      "started_at": "2026-10-02T08:42:56.952333+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 20,
            "unmarked": 22
          },
          "event": "at_least_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is at least one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.77,
      "valid": true,
      "error": null,
      "cost_usd": 1.7682e-05,
      "usage": {
        "input_tokens": 421,
        "output_tokens": 20,
        "cost": 1.7682e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.77
          }
        },
        "usage": {
          "input_tokens": 421,
          "output_tokens": 20,
          "cost": 1.7682e-05
        },
        "id": "gen-dec-1790930576-50GZcXvtJOiA8i2FmghD",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930576-50GZcXvtJOiA8i2FmghD",
      "answer": {
        "type": "noul",
        "noul": 0.77
      },
      "latency_s": 0.22428966700681485,
      "run": "main",
      "record_index": 48
    },
    {
      "case_id": "probability-018",
      "experiment": "probability",
      "target": 0.7317073170731707,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 49,
      "started_at": "2026-10-02T08:42:57.177898+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is at least one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 20, \"unmarked\": 22}, \"event\": \"at_least_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.7585,
      "valid": true,
      "error": null,
      "cost_usd": 3.97e-05,
      "usage": {
        "prompt_tokens": 282,
        "completion_tokens": 23,
        "total_tokens": 305,
        "cost": 3.97e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.97e-05,
          "upstream_inference_prompt_cost": 2.82e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930577-GtCirtsM6q9lxxQQtw39",
        "object": "chat.completion",
        "created": 1790930577,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.7585}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 282,
          "completion_tokens": 23,
          "total_tokens": 305,
          "cost": 3.97e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.97e-05,
            "upstream_inference_prompt_cost": 2.82e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930577-GtCirtsM6q9lxxQQtw39",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.7585
      },
      "latency_s": 1.4449781250150409,
      "run": "main",
      "record_index": 49
    },
    {
      "case_id": "boolq-dev-0055",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 50,
      "started_at": "2026-10-02T08:42:58.623404+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "In the third season, Damon helps Elena in bringing his brother, Stefan, back to Mystic Falls after Stefan becomes Klaus' henchman. The arrangement transpired after a bargain for his blood that would cure Damon of the werewolf bite he had received from Tyler. At first, he is reluctant to involve Elena in the rescue attempts, employing Alaric Saltzman, Elena's guardian, instead as Klaus does not know that Elena is alive after the sacrifice which frees Klaus' hybrid side. However, Elena involves herself, desperate to find Stefan. Damon, though hesitant at first, is unable to refuse her because of his love for her. He also points out to her that she once turned back from finding Stefan since she knew Damon would be in danger, clearly showing that she also has feelings for him. He tells her that ``when (he) drag(s) (his) brother from the edge to deliver him back to (her), (he) wants her to remember the things (she) felt while he was gone.'' When Stefan finally returns to Mystic Falls, his attitude is different from that of the first and second seasons. This causes a rift between Elena and Stefan whereas the relationship between Damon and Elena becomes closer and more intimate. A still loyal Elena, however, refuses to admit her feelings for Damon. In 'Dangerous Liaisons', Elena, frustrated with her feelings for him, tells Damon that his love for her may be a problem, and that this could be causing all their troubles. This incenses Damon, causing him to revert to the uncaring and reckless Damon seen in the previous seasons. The rocky relationship between the two continues until the sexual tension hits the fan and in a moment of heated passion, Elena -- for the first time in the three seasons -- kisses Damon of her own accord. This kiss finally causes Elena to admit that she loves both brothers and realize that she must ultimately make her choice as her own ancestress, Katherine Pierce, who turned the brothers, once did. In assessment of her feelings for Damon, she states this: ``Damon just sort of snuck up on me. He got under my skin and no matter what I do, I can't shake him.'' In the season finale, a trip designed to get her to safety forces Elena to make her choice: to go to Damon and possibly see him one last time; or to go to Stefan and her friends and see them one last time. She chooses the latter when she calls Damon to tell him her decision. Damon, who is trying to stop Alaric, accepts what she says and she tells him that maybe if she had met Damon before she had met Stefan, her choice may have been different. This statement causes Damon to remember the first night he did meet Elena which was, in fact, the night her parents died - before she had met Stefan. Not wanting anyone to know he was in town and after giving her some advice about life and love, Damon compels her to forget. He remembers this as he fights Alaric and seems accepting of his death when Alaric, whose life line is tied to Elena's, suddenly collapses in his arms. Damon is grief-stricken, knowing that this means that Elena has also died and yells, ``No! You are not dead!'' A heartbroken Damon then goes to the hospital demanding to see Elena when the doctor, Meredith Fell, tells him that she gave Elena vampire blood. The last shot of the season finale episode shows Elena in transition."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "does damon and elena get together in season 3?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.38,
      "valid": true,
      "error": null,
      "cost_usd": 4.4016e-05,
      "usage": {
        "input_tokens": 1048,
        "output_tokens": 20,
        "cost": 4.4016e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.38
          }
        },
        "usage": {
          "input_tokens": 1048,
          "output_tokens": 20,
          "cost": 4.4016e-05
        },
        "id": "gen-dec-1790930578-NVxlprZentB7rdCZww1H",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930578-NVxlprZentB7rdCZww1H",
      "answer": {
        "type": "noul",
        "noul": 0.38
      },
      "latency_s": 0.251548542000819,
      "run": "main",
      "record_index": 50
    },
    {
      "case_id": "boolq-dev-0055",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 51,
      "started_at": "2026-10-02T08:42:58.875420+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"does damon and elena get together in season 3?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"In the third season, Damon helps Elena in bringing his brother, Stefan, back to Mystic Falls after Stefan becomes Klaus' henchman. The arrangement transpired after a bargain for his blood that would cure Damon of the werewolf bite he had received from Tyler. At first, he is reluctant to involve Elena in the rescue attempts, employing Alaric Saltzman, Elena's guardian, instead as Klaus does not know that Elena is alive after the sacrifice which frees Klaus' hybrid side. However, Elena involves herself, desperate to find Stefan. Damon, though hesitant at first, is unable to refuse her because of his love for her. He also points out to her that she once turned back from finding Stefan since she knew Damon would be in danger, clearly showing that she also has feelings for him. He tells her that ``when (he) drag(s) (his) brother from the edge to deliver him back to (her), (he) wants her to remember the things (she) felt while he was gone.'' When Stefan finally returns to Mystic Falls, his attitude is different from that of the first and second seasons. This causes a rift between Elena and Stefan whereas the relationship between Damon and Elena becomes closer and more intimate. A still loyal Elena, however, refuses to admit her feelings for Damon. In 'Dangerous Liaisons', Elena, frustrated with her feelings for him, tells Damon that his love for her may be a problem, and that this could be causing all their troubles. This incenses Damon, causing him to revert to the uncaring and reckless Damon seen in the previous seasons. The rocky relationship between the two continues until the sexual tension hits the fan and in a moment of heated passion, Elena -- for the first time in the three seasons -- kisses Damon of her own accord. This kiss finally causes Elena to admit that she loves both brothers and realize that she must ultimately make her choice as her own ancestress, Katherine Pierce, who turned the brothers, once did. In assessment of her feelings for Damon, she states this: ``Damon just sort of snuck up on me. He got under my skin and no matter what I do, I can't shake him.'' In the season finale, a trip designed to get her to safety forces Elena to make her choice: to go to Damon and possibly see him one last time; or to go to Stefan and her friends and see them one last time. She chooses the latter when she calls Damon to tell him her decision. Damon, who is trying to stop Alaric, accepts what she says and she tells him that maybe if she had met Damon before she had met Stefan, her choice may have been different. This statement causes Damon to remember the first night he did meet Elena which was, in fact, the night her parents died - before she had met Stefan. Not wanting anyone to know he was in town and after giving her some advice about life and love, Damon compels her to forget. He remembers this as he fights Alaric and seems accepting of his death when Alaric, whose life line is tied to Elena's, suddenly collapses in his arms. Damon is grief-stricken, knowing that this means that Elena has also died and yells, ``No! You are not dead!'' A heartbroken Damon then goes to the hospital demanding to see Elena when the doctor, Meredith Fell, tells him that she gave Elena vampire blood. The last shot of the season finale episode shows Elena in transition.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.15,
      "valid": true,
      "error": null,
      "cost_usd": 0.0001029,
      "usage": {
        "prompt_tokens": 919,
        "completion_tokens": 22,
        "total_tokens": 941,
        "cost": 0.0001029,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0001029,
          "upstream_inference_prompt_cost": 9.19e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930578-Byp7qbzTbHOTxCE78N9k",
        "object": "chat.completion",
        "created": 1790930578,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.15}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 919,
          "completion_tokens": 22,
          "total_tokens": 941,
          "cost": 0.0001029,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0001029,
            "upstream_inference_prompt_cost": 9.19e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930578-Byp7qbzTbHOTxCE78N9k",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.15
      },
      "latency_s": 1.0378335420100484,
      "run": "main",
      "record_index": 51
    },
    {
      "case_id": "boolq-dev-2504",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 52,
      "started_at": "2026-10-02T08:42:59.913717+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Ian McKellen was offered the role, but he turned it down, having played the similar character Gandalf in The Lord of the Rings trilogy, as well as feeling it would have been inappropriate to take Harris's role, as Harris had called McKellen a ``dreadful'' actor. Harris's family had expressed an interest in seeing Peter O'Toole being chosen as his replacement."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "did the actor who played gandalf play dumbledore?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.23,
      "valid": true,
      "error": null,
      "cost_usd": 1.7346e-05,
      "usage": {
        "input_tokens": 413,
        "output_tokens": 20,
        "cost": 1.7346e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.23
          }
        },
        "usage": {
          "input_tokens": 413,
          "output_tokens": 20,
          "cost": 1.7346e-05
        },
        "id": "gen-dec-1790930579-qoFNPh8KbjRnG3Gh5d3c",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930579-qoFNPh8KbjRnG3Gh5d3c",
      "answer": {
        "type": "noul",
        "noul": 0.23
      },
      "latency_s": 0.24152725000749342,
      "run": "main",
      "record_index": 52
    },
    {
      "case_id": "boolq-dev-2504",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 53,
      "started_at": "2026-10-02T08:43:00.155804+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"did the actor who played gandalf play dumbledore?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Ian McKellen was offered the role, but he turned it down, having played the similar character Gandalf in The Lord of the Rings trilogy, as well as feeling it would have been inappropriate to take Harris's role, as Harris had called McKellen a ``dreadful'' actor. Harris's family had expressed an interest in seeing Peter O'Toole being chosen as his replacement.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 3.98e-05,
      "usage": {
        "prompt_tokens": 288,
        "completion_tokens": 22,
        "total_tokens": 310,
        "cost": 3.98e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.98e-05,
          "upstream_inference_prompt_cost": 2.88e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930580-5K6ynnYCObeobVCvnDIs",
        "object": "chat.completion",
        "created": 1790930580,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.98}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 288,
          "completion_tokens": 22,
          "total_tokens": 310,
          "cost": 3.98e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.98e-05,
            "upstream_inference_prompt_cost": 2.88e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930580-5K6ynnYCObeobVCvnDIs",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 1.0501592080108821,
      "run": "main",
      "record_index": 53
    },
    {
      "case_id": "probability-039",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 54,
      "started_at": "2026-10-02T08:43:01.206618+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 0,
              "unmarked": 16
            },
            "B": {
              "marked": 12,
              "unmarked": 48
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.23,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.23
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930581-da0q9jD9o3XRNbPemF7q",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930581-da0q9jD9o3XRNbPemF7q",
      "answer": {
        "type": "noul",
        "noul": 0.23
      },
      "latency_s": 0.2223374169843737,
      "run": "main",
      "record_index": 54
    },
    {
      "case_id": "probability-039",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 55,
      "started_at": "2026-10-02T08:43:01.429569+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 0, \"unmarked\": 16}, \"B\": {\"marked\": 12, \"unmarked\": 48}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.2,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930581-WAPihBFuf2J5XvHBYzHZ",
        "object": "chat.completion",
        "created": 1790930581,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.2}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930581-WAPihBFuf2J5XvHBYzHZ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.2
      },
      "latency_s": 1.079251999995904,
      "run": "main",
      "record_index": 55
    },
    {
      "case_id": "probability-055",
      "experiment": "probability",
      "target": 0.7051282051282052,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 56,
      "started_at": "2026-10-02T08:43:02.509228+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 11,
            "unmarked": 2
          },
          "event": "both_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Are both selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.72,
      "valid": true,
      "error": null,
      "cost_usd": 1.7304e-05,
      "usage": {
        "input_tokens": 412,
        "output_tokens": 20,
        "cost": 1.7304e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.72
          }
        },
        "usage": {
          "input_tokens": 412,
          "output_tokens": 20,
          "cost": 1.7304e-05
        },
        "id": "gen-dec-1790930582-L3LigACaBemNf4R3VUev",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930582-L3LigACaBemNf4R3VUev",
      "answer": {
        "type": "noul",
        "noul": 0.72
      },
      "latency_s": 0.2618421249790117,
      "run": "main",
      "record_index": 56
    },
    {
      "case_id": "probability-055",
      "experiment": "probability",
      "target": 0.7051282051282052,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 57,
      "started_at": "2026-10-02T08:43:02.771589+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Are both selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 11, \"unmarked\": 2}, \"event\": \"both_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.7051282051,
      "valid": true,
      "error": null,
      "cost_usd": 3.99e-05,
      "usage": {
        "prompt_tokens": 274,
        "completion_tokens": 25,
        "total_tokens": 299,
        "cost": 3.99e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.99e-05,
          "upstream_inference_prompt_cost": 2.74e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930582-B7GU9G6MvwNacp7T74cY",
        "object": "chat.completion",
        "created": 1790930582,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.7051282051}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 274,
          "completion_tokens": 25,
          "total_tokens": 299,
          "cost": 3.99e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.99e-05,
            "upstream_inference_prompt_cost": 2.74e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930582-B7GU9G6MvwNacp7T74cY",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.7051282051
      },
      "latency_s": 1.041061833006097,
      "run": "main",
      "record_index": 57
    },
    {
      "case_id": "policy-013",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 58,
      "started_at": "2026-10-02T08:43:03.813437+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 91, \"defective\": true, \"final_sale\": true, \"price_credits\": 334, \"recalled\": true, \"receipt\": false, \"unused\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930583-y6inR4pmv1jpgfhUZevJ",
        "object": "chat.completion",
        "created": 1790930583,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930583-y6inR4pmv1jpgfhUZevJ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0172764169983566,
      "run": "main",
      "record_index": 58
    },
    {
      "case_id": "policy-013",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 59,
      "started_at": "2026-10-02T08:43:04.831351+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 91,
            "defective": true,
            "final_sale": true,
            "price_credits": 334,
            "recalled": true,
            "receipt": false,
            "unused": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930584-bmPZopEE2GHwQc7ETcWw",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930584-bmPZopEE2GHwQc7ETcWw",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.2217986669857055,
      "run": "main",
      "record_index": 59
    },
    {
      "case_id": "policy-012",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 60,
      "started_at": "2026-10-02T08:43:05.053558+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "green",
            "clearance": 4,
            "emergency_permit": true,
            "experience_years": 10,
            "role": "contractor",
            "sponsor": false,
            "suspended": false,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1126e-05,
      "usage": {
        "input_tokens": 503,
        "output_tokens": 20,
        "cost": 2.1126e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 503,
          "output_tokens": 20,
          "cost": 2.1126e-05
        },
        "id": "gen-dec-1790930585-iDqacHVwqvzyXpnV6X2O",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930585-iDqacHVwqvzyXpnV6X2O",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.233893166994676,
      "run": "main",
      "record_index": 60
    },
    {
      "case_id": "policy-012",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 61,
      "started_at": "2026-10-02T08:43:05.287972+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"green\", \"clearance\": 4, \"emergency_permit\": true, \"experience_years\": 10, \"role\": \"contractor\", \"sponsor\": false, \"suspended\": false, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.62e-05,
      "usage": {
        "prompt_tokens": 362,
        "completion_tokens": 20,
        "total_tokens": 382,
        "cost": 4.62e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.62e-05,
          "upstream_inference_prompt_cost": 3.62e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930585-GGEeK1fLOPpnyu4BzpCp",
        "object": "chat.completion",
        "created": 1790930585,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 362,
          "completion_tokens": 20,
          "total_tokens": 382,
          "cost": 4.62e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.62e-05,
            "upstream_inference_prompt_cost": 3.62e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930585-GGEeK1fLOPpnyu4BzpCp",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.2701927089947276,
      "run": "main",
      "record_index": 61
    },
    {
      "case_id": "policy-025",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 62,
      "started_at": "2026-10-02T08:43:06.558828+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 320,
            "active_promotion": true,
            "gift_wrapping": false,
            "hazardous": true,
            "mainland": false,
            "member": true,
            "subtotal_credits": 101,
            "weight_kg": 21
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.07,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.07
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930586-Z7wsPlyJhVFC9W1qZ7WS",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930586-Z7wsPlyJhVFC9W1qZ7WS",
      "answer": {
        "type": "noul",
        "noul": 0.07
      },
      "latency_s": 0.25420595897594467,
      "run": "main",
      "record_index": 62
    },
    {
      "case_id": "policy-025",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 63,
      "started_at": "2026-10-02T08:43:06.813487+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 320, \"active_promotion\": true, \"gift_wrapping\": false, \"hazardous\": true, \"mainland\": false, \"member\": true, \"subtotal_credits\": 101, \"weight_kg\": 21}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.64e-05,
      "usage": {
        "prompt_tokens": 364,
        "completion_tokens": 20,
        "total_tokens": 384,
        "cost": 4.64e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.64e-05,
          "upstream_inference_prompt_cost": 3.64e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930586-OhcenZMYTRtgR7efhcwL",
        "object": "chat.completion",
        "created": 1790930586,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 364,
          "completion_tokens": 20,
          "total_tokens": 384,
          "cost": 4.64e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.64e-05,
            "upstream_inference_prompt_cost": 3.64e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930586-OhcenZMYTRtgR7efhcwL",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.1727671659900807,
      "run": "main",
      "record_index": 63
    },
    {
      "case_id": "policy-047",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 64,
      "started_at": "2026-10-02T08:43:07.986736+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "orange",
            "clearance": 1,
            "emergency_permit": false,
            "experience_years": 29,
            "role": "visitor",
            "sponsor": true,
            "suspended": true,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.04,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.04
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930588-1plFFHWDdZjhdtL6a6xR",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930588-1plFFHWDdZjhdtL6a6xR",
      "answer": {
        "type": "noul",
        "noul": 0.04
      },
      "latency_s": 0.25377937499433756,
      "run": "main",
      "record_index": 64
    },
    {
      "case_id": "policy-047",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 65,
      "started_at": "2026-10-02T08:43:08.240917+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"orange\", \"clearance\": 1, \"emergency_permit\": false, \"experience_years\": 29, \"role\": \"visitor\", \"sponsor\": true, \"suspended\": true, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930588-wmycTshCFUfSqPQZfl2c",
        "object": "chat.completion",
        "created": 1790930588,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930588-wmycTshCFUfSqPQZfl2c",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9818080409895629,
      "run": "main",
      "record_index": 65
    },
    {
      "case_id": "policy-041",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 66,
      "started_at": "2026-10-02T08:43:09.223155+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": false,
            "badge_color": "orange",
            "clearance": 2,
            "emergency_permit": true,
            "experience_years": 8,
            "role": "visitor",
            "sponsor": false,
            "suspended": false,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.23,
      "valid": true,
      "error": null,
      "cost_usd": 2.1042e-05,
      "usage": {
        "input_tokens": 501,
        "output_tokens": 20,
        "cost": 2.1042e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.23
          }
        },
        "usage": {
          "input_tokens": 501,
          "output_tokens": 20,
          "cost": 2.1042e-05
        },
        "id": "gen-dec-1790930589-4ETG4IaE9WXaexxWhMZP",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930589-4ETG4IaE9WXaexxWhMZP",
      "answer": {
        "type": "noul",
        "noul": 0.23
      },
      "latency_s": 0.25552483298815787,
      "run": "main",
      "record_index": 66
    },
    {
      "case_id": "policy-041",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 67,
      "started_at": "2026-10-02T08:43:09.479207+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": false, \"badge_color\": \"orange\", \"clearance\": 2, \"emergency_permit\": true, \"experience_years\": 8, \"role\": \"visitor\", \"sponsor\": false, \"suspended\": false, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930589-tN3GSGEuECdil6ftJtFH",
        "object": "chat.completion",
        "created": 1790930589,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930589-tN3GSGEuECdil6ftJtFH",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9898341250082012,
      "run": "main",
      "record_index": 67
    },
    {
      "case_id": "boolq-dev-0448",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 68,
      "started_at": "2026-10-02T08:43:10.469528+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Sting later said he was disconcerted by how many people think the song is more positive than it is. He insists it is about the obsession with a lost lover, and the jealousy and surveillance that follow. ``One couple told me 'Oh we love that song; it was the main song played at our wedding!' I thought, 'Well, good luck.''' When asked why he appears angry in the music video, Sting told BBC Radio 2, ``I think the song is very, very sinister and ugly and people have actually misinterpreted it as being a gentle little love song, when it's quite the opposite.''"
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is every step you take about a stalker?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.61,
      "valid": true,
      "error": null,
      "cost_usd": 1.9278e-05,
      "usage": {
        "input_tokens": 459,
        "output_tokens": 20,
        "cost": 1.9278e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.61
          }
        },
        "usage": {
          "input_tokens": 459,
          "output_tokens": 20,
          "cost": 1.9278e-05
        },
        "id": "gen-dec-1790930590-4posaiqz6WDNh7EP74Xv",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930590-4posaiqz6WDNh7EP74Xv",
      "answer": {
        "type": "noul",
        "noul": 0.61
      },
      "latency_s": 0.22120529200765304,
      "run": "main",
      "record_index": 68
    },
    {
      "case_id": "boolq-dev-0448",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 69,
      "started_at": "2026-10-02T08:43:10.691135+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is every step you take about a stalker?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Sting later said he was disconcerted by how many people think the song is more positive than it is. He insists it is about the obsession with a lost lover, and the jealousy and surveillance that follow. ``One couple told me 'Oh we love that song; it was the main song played at our wedding!' I thought, 'Well, good luck.''' When asked why he appears angry in the music video, Sting told BBC Radio 2, ``I think the song is very, very sinister and ugly and people have actually misinterpreted it as being a gentle little love song, when it's quite the opposite.''\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.33e-05,
      "usage": {
        "prompt_tokens": 333,
        "completion_tokens": 20,
        "total_tokens": 353,
        "cost": 4.33e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.33e-05,
          "upstream_inference_prompt_cost": 3.33e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930590-vArnlW3Y9brq9y2wESks",
        "object": "chat.completion",
        "created": 1790930590,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 333,
          "completion_tokens": 20,
          "total_tokens": 353,
          "cost": 4.33e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.33e-05,
            "upstream_inference_prompt_cost": 3.33e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930590-vArnlW3Y9brq9y2wESks",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9778644590114709,
      "run": "main",
      "record_index": 69
    },
    {
      "case_id": "probability-042",
      "experiment": "probability",
      "target": 0.0,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 70,
      "started_at": "2026-10-02T08:43:11.669791+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is exactly one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 0, \"unmarked\": 18}, \"event\": \"exactly_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.8e-05,
      "usage": {
        "prompt_tokens": 280,
        "completion_tokens": 20,
        "total_tokens": 300,
        "cost": 3.8e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.8e-05,
          "upstream_inference_prompt_cost": 2.8e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930591-OrT7TTQWF72OHZPrdhHD",
        "object": "chat.completion",
        "created": 1790930591,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 280,
          "completion_tokens": 20,
          "total_tokens": 300,
          "cost": 3.8e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.8e-05,
            "upstream_inference_prompt_cost": 2.8e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930591-OrT7TTQWF72OHZPrdhHD",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2764361249865033,
      "run": "main",
      "record_index": 70
    },
    {
      "case_id": "probability-042",
      "experiment": "probability",
      "target": 0.0,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 71,
      "started_at": "2026-10-02T08:43:12.948338+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 0,
            "unmarked": 18
          },
          "event": "exactly_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is exactly one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.03,
      "valid": true,
      "error": null,
      "cost_usd": 1.7598e-05,
      "usage": {
        "input_tokens": 419,
        "output_tokens": 20,
        "cost": 1.7598e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.03
          }
        },
        "usage": {
          "input_tokens": 419,
          "output_tokens": 20,
          "cost": 1.7598e-05
        },
        "id": "gen-dec-1790930592-wbJ2kUaENM4OjLFhsJre",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930592-wbJ2kUaENM4OjLFhsJre",
      "answer": {
        "type": "noul",
        "noul": 0.03
      },
      "latency_s": 0.22932975000003353,
      "run": "main",
      "record_index": 71
    },
    {
      "case_id": "policy-039",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 72,
      "started_at": "2026-10-02T08:43:13.179350+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 1700,
            "active_promotion": true,
            "gift_wrapping": false,
            "hazardous": false,
            "mainland": true,
            "member": true,
            "subtotal_credits": 49,
            "weight_kg": 19
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930593-LYtXFa7hnwLx3EBoYnUR",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930593-LYtXFa7hnwLx3EBoYnUR",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.22608329201466404,
      "run": "main",
      "record_index": 72
    },
    {
      "case_id": "policy-039",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 73,
      "started_at": "2026-10-02T08:43:13.407300+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 1700, \"active_promotion\": true, \"gift_wrapping\": false, \"hazardous\": false, \"mainland\": true, \"member\": true, \"subtotal_credits\": 49, \"weight_kg\": 19}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930593-KYQmHHTR2TmacgyTnYQr",
        "object": "chat.completion",
        "created": 1790930593,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930593-KYQmHHTR2TmacgyTnYQr",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.5924109590123408,
      "run": "main",
      "record_index": 73
    },
    {
      "case_id": "policy-036",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 74,
      "started_at": "2026-10-02T08:43:15.000251+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": false, \"days_since_purchase\": 91, \"defective\": true, \"final_sale\": false, \"price_credits\": 31, \"recalled\": false, \"receipt\": true, \"unused\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930595-3a7U75yzHAu6cHpXnY10",
        "object": "chat.completion",
        "created": 1790930595,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930595-3a7U75yzHAu6cHpXnY10",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0032162909919862,
      "run": "main",
      "record_index": 74
    },
    {
      "case_id": "policy-036",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 75,
      "started_at": "2026-10-02T08:43:16.003909+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": false,
            "days_since_purchase": 91,
            "defective": true,
            "final_sale": false,
            "price_credits": 31,
            "recalled": false,
            "receipt": true,
            "unused": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 2.037e-05,
      "usage": {
        "input_tokens": 485,
        "output_tokens": 20,
        "cost": 2.037e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 485,
          "output_tokens": 20,
          "cost": 2.037e-05
        },
        "id": "gen-dec-1790930596-Xj2bko2RqIs0N1SbaZjs",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930596-Xj2bko2RqIs0N1SbaZjs",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.26285408399417065,
      "run": "main",
      "record_index": 75
    },
    {
      "case_id": "boolq-dev-1625",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 76,
      "started_at": "2026-10-02T08:43:16.267170+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Black garlic can be eaten alone, on bread, or used in soups, sauces, crushed into a mayonnaise or simply tossed into a vegetable dish. A vinaigrette can be made with black garlic, sherry vinegar, soy, a neutral oil, and Dijon mustard. Its softness increases with water content."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is japanese black garlic supposed to be mushy?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.66,
      "valid": true,
      "error": null,
      "cost_usd": 1.659e-05,
      "usage": {
        "input_tokens": 395,
        "output_tokens": 20,
        "cost": 1.659e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.66
          }
        },
        "usage": {
          "input_tokens": 395,
          "output_tokens": 20,
          "cost": 1.659e-05
        },
        "id": "gen-dec-1790930596-aOH0XpsEEVbGahnZap0I",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930596-aOH0XpsEEVbGahnZap0I",
      "answer": {
        "type": "noul",
        "noul": 0.66
      },
      "latency_s": 0.22163020799052902,
      "run": "main",
      "record_index": 76
    },
    {
      "case_id": "boolq-dev-1625",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 77,
      "started_at": "2026-10-02T08:43:16.489314+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is japanese black garlic supposed to be mushy?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Black garlic can be eaten alone, on bread, or used in soups, sauces, crushed into a mayonnaise or simply tossed into a vegetable dish. A vinaigrette can be made with black garlic, sherry vinegar, soy, a neutral oil, and Dijon mustard. Its softness increases with water content.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.25,
      "valid": true,
      "error": null,
      "cost_usd": 3.77e-05,
      "usage": {
        "prompt_tokens": 267,
        "completion_tokens": 22,
        "total_tokens": 289,
        "cost": 3.77e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.77e-05,
          "upstream_inference_prompt_cost": 2.67e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930596-Mt7PqjKeKIkGDkDALm1d",
        "object": "chat.completion",
        "created": 1790930596,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.25}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 267,
          "completion_tokens": 22,
          "total_tokens": 289,
          "cost": 3.77e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.77e-05,
            "upstream_inference_prompt_cost": 2.67e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930596-Mt7PqjKeKIkGDkDALm1d",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.25
      },
      "latency_s": 1.2601647079864051,
      "run": "main",
      "record_index": 77
    },
    {
      "case_id": "policy-040",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 78,
      "started_at": "2026-10-02T08:43:17.750003+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 2869,
            "active_promotion": false,
            "gift_wrapping": false,
            "hazardous": false,
            "mainland": true,
            "member": true,
            "subtotal_credits": 50,
            "weight_kg": 19
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.97,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.97
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930597-y8mm5HWbsIdExya9DByw",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930597-y8mm5HWbsIdExya9DByw",
      "answer": {
        "type": "noul",
        "noul": 0.97
      },
      "latency_s": 0.22932874999241903,
      "run": "main",
      "record_index": 78
    },
    {
      "case_id": "policy-040",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 79,
      "started_at": "2026-10-02T08:43:17.979808+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 2869, \"active_promotion\": false, \"gift_wrapping\": false, \"hazardous\": false, \"mainland\": true, \"member\": true, \"subtotal_credits\": 50, \"weight_kg\": 19}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930597-zrNJeVbA8kTzZq0kGeai",
        "object": "chat.completion",
        "created": 1790930597,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930597-zrNJeVbA8kTzZq0kGeai",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9506033340003341,
      "run": "main",
      "record_index": 79
    },
    {
      "case_id": "policy-001",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 80,
      "started_at": "2026-10-02T08:43:18.931430+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "orange",
            "clearance": 2,
            "emergency_permit": true,
            "experience_years": 16,
            "role": "contractor",
            "sponsor": false,
            "suspended": false,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1126e-05,
      "usage": {
        "input_tokens": 503,
        "output_tokens": 20,
        "cost": 2.1126e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 503,
          "output_tokens": 20,
          "cost": 2.1126e-05
        },
        "id": "gen-dec-1790930598-vqtKB6CuPTv1uf9TSKjt",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930598-vqtKB6CuPTv1uf9TSKjt",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.27906433399766684,
      "run": "main",
      "record_index": 80
    },
    {
      "case_id": "policy-001",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 81,
      "started_at": "2026-10-02T08:43:19.210904+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"orange\", \"clearance\": 2, \"emergency_permit\": true, \"experience_years\": 16, \"role\": \"contractor\", \"sponsor\": false, \"suspended\": false, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.62e-05,
      "usage": {
        "prompt_tokens": 362,
        "completion_tokens": 20,
        "total_tokens": 382,
        "cost": 4.62e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.62e-05,
          "upstream_inference_prompt_cost": 3.62e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930599-hJD7stb5lzKiDtPRl0PK",
        "object": "chat.completion",
        "created": 1790930599,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 362,
          "completion_tokens": 20,
          "total_tokens": 382,
          "cost": 4.62e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.62e-05,
            "upstream_inference_prompt_cost": 3.62e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930599-hJD7stb5lzKiDtPRl0PK",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0651056250208057,
      "run": "main",
      "record_index": 81
    },
    {
      "case_id": "policy-028",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 82,
      "started_at": "2026-10-02T08:43:20.276566+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": false,
            "badge_color": "orange",
            "clearance": 3,
            "emergency_permit": true,
            "experience_years": 22,
            "role": "visitor",
            "sponsor": true,
            "suspended": false,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.2,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.2
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930600-RrkCAtwINs9Q7JTTCgMk",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930600-RrkCAtwINs9Q7JTTCgMk",
      "answer": {
        "type": "noul",
        "noul": 0.2
      },
      "latency_s": 0.4001794579962734,
      "run": "main",
      "record_index": 82
    },
    {
      "case_id": "policy-028",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 83,
      "started_at": "2026-10-02T08:43:20.677255+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": false, \"badge_color\": \"orange\", \"clearance\": 3, \"emergency_permit\": true, \"experience_years\": 22, \"role\": \"visitor\", \"sponsor\": true, \"suspended\": false, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930600-NS2oVFtxZFVwgPONa0pJ",
        "object": "chat.completion",
        "created": 1790930600,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930600-NS2oVFtxZFVwgPONa0pJ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.4985671250033192,
      "run": "main",
      "record_index": 83
    },
    {
      "case_id": "probability-014",
      "experiment": "probability",
      "target": 0.95,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 84,
      "started_at": "2026-10-02T08:43:22.176224+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 37,
              "unmarked": 14
            },
            "B": {
              "marked": 19,
              "unmarked": 1
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.93,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.93
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930602-xkgOglFSBwNaGLxxEdUH",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930602-xkgOglFSBwNaGLxxEdUH",
      "answer": {
        "type": "noul",
        "noul": 0.93
      },
      "latency_s": 0.25286516701453365,
      "run": "main",
      "record_index": 84
    },
    {
      "case_id": "probability-014",
      "experiment": "probability",
      "target": 0.95,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 85,
      "started_at": "2026-10-02T08:43:22.429543+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 37, \"unmarked\": 14}, \"B\": {\"marked\": 19, \"unmarked\": 1}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.95,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930602-Li8Bq7BcB0h0esGGGKpO",
        "object": "chat.completion",
        "created": 1790930602,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.95}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930602-Li8Bq7BcB0h0esGGGKpO",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.95
      },
      "latency_s": 1.43668612500187,
      "run": "main",
      "record_index": 85
    },
    {
      "case_id": "boolq-dev-3009",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 86,
      "started_at": "2026-10-02T08:43:23.866715+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The New York City Subway is a rapid transit system owned by the City of New York and leased to the New York City Transit Authority, a subsidiary agency of the state-run Metropolitan Transportation Authority (MTA). Opened in 1904, the New York City Subway is one of the world's oldest public transit systems, one of the world's most used metro systems, and the metro system with the most stations. It offers service 24 hours per day on every day of the year, though some routes may operate only part-time."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "does the subway in nyc run 24 hours?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.94,
      "valid": true,
      "error": null,
      "cost_usd": 1.8564e-05,
      "usage": {
        "input_tokens": 442,
        "output_tokens": 20,
        "cost": 1.8564e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.94
          }
        },
        "usage": {
          "input_tokens": 442,
          "output_tokens": 20,
          "cost": 1.8564e-05
        },
        "id": "gen-dec-1790930603-cJaG9WUURQtMoOak4CpX",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930603-cJaG9WUURQtMoOak4CpX",
      "answer": {
        "type": "noul",
        "noul": 0.94
      },
      "latency_s": 0.29438295899308287,
      "run": "main",
      "record_index": 86
    },
    {
      "case_id": "boolq-dev-3009",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 87,
      "started_at": "2026-10-02T08:43:24.163311+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"does the subway in nyc run 24 hours?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The New York City Subway is a rapid transit system owned by the City of New York and leased to the New York City Transit Authority, a subsidiary agency of the state-run Metropolitan Transportation Authority (MTA). Opened in 1904, the New York City Subway is one of the world's oldest public transit systems, one of the world's most used metro systems, and the metro system with the most stations. It offers service 24 hours per day on every day of the year, though some routes may operate only part-time.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.12e-05,
      "usage": {
        "prompt_tokens": 312,
        "completion_tokens": 20,
        "total_tokens": 332,
        "cost": 4.12e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.12e-05,
          "upstream_inference_prompt_cost": 3.12e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930604-aaSTxnhlP7JxlL6VqB8A",
        "object": "chat.completion",
        "created": 1790930604,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 312,
          "completion_tokens": 20,
          "total_tokens": 332,
          "cost": 4.12e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.12e-05,
            "upstream_inference_prompt_cost": 3.12e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930604-aaSTxnhlP7JxlL6VqB8A",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0349687919951975,
      "run": "main",
      "record_index": 87
    },
    {
      "case_id": "boolq-dev-0341",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 88,
      "started_at": "2026-10-02T08:43:25.198742+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "In common everyday speech, speed of sound refers to the speed of sound waves in air. However, the speed of sound varies from substance to substance: sound travels most slowly in gases; it travels faster in liquids; and faster still in solids. For example, (as noted above), sound travels at 343 m/s in air; it travels at 1,484 m/s in water (4.3 times as fast as in air); and at 5,120 m/s in iron (about 15 times as fast as in air). In an exceptionally stiff material such as diamond, sound travels at 12,000 metres per second (26,843 mph); (about 35 times as fast as in air) which is around the maximum speed that sound will travel under normal conditions."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is the speed of sound constant for all materials?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.03,
      "valid": true,
      "error": null,
      "cost_usd": 2.1126e-05,
      "usage": {
        "input_tokens": 503,
        "output_tokens": 20,
        "cost": 2.1126e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.03
          }
        },
        "usage": {
          "input_tokens": 503,
          "output_tokens": 20,
          "cost": 2.1126e-05
        },
        "id": "gen-dec-1790930605-UlK0iJHiBfB2wR8uQCW0",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930605-UlK0iJHiBfB2wR8uQCW0",
      "answer": {
        "type": "noul",
        "noul": 0.03
      },
      "latency_s": 0.23289520799880847,
      "run": "main",
      "record_index": 88
    },
    {
      "case_id": "boolq-dev-0341",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 89,
      "started_at": "2026-10-02T08:43:25.432064+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is the speed of sound constant for all materials?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"In common everyday speech, speed of sound refers to the speed of sound waves in air. However, the speed of sound varies from substance to substance: sound travels most slowly in gases; it travels faster in liquids; and faster still in solids. For example, (as noted above), sound travels at 343 m/s in air; it travels at 1,484 m/s in water (4.3 times as fast as in air); and at 5,120 m/s in iron (about 15 times as fast as in air). In an exceptionally stiff material such as diamond, sound travels at 12,000 metres per second (26,843 mph); (about 35 times as fast as in air) which is around the maximum speed that sound will travel under normal conditions.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.66e-05,
      "usage": {
        "prompt_tokens": 366,
        "completion_tokens": 20,
        "total_tokens": 386,
        "cost": 4.66e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.66e-05,
          "upstream_inference_prompt_cost": 3.66e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930605-9ADhSkkNmIyyL6Dj4X0H",
        "object": "chat.completion",
        "created": 1790930605,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 366,
          "completion_tokens": 20,
          "total_tokens": 386,
          "cost": 4.66e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.66e-05,
            "upstream_inference_prompt_cost": 3.66e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930605-9ADhSkkNmIyyL6Dj4X0H",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.478363125002943,
      "run": "main",
      "record_index": 89
    },
    {
      "case_id": "probability-020",
      "experiment": "probability",
      "target": 0.5458937198067633,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 90,
      "started_at": "2026-10-02T08:43:26.910958+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 1,
            "B": 2
          },
          "boxes": {
            "A": {
              "marked": 14,
              "unmarked": 32
            },
            "B": {
              "marked": 16,
              "unmarked": 8
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.38,
      "valid": true,
      "error": null,
      "cost_usd": 2.0034e-05,
      "usage": {
        "input_tokens": 477,
        "output_tokens": 20,
        "cost": 2.0034e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.38
          }
        },
        "usage": {
          "input_tokens": 477,
          "output_tokens": 20,
          "cost": 2.0034e-05
        },
        "id": "gen-dec-1790930606-rMY6O0WihCxMafG8f3OB",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930606-rMY6O0WihCxMafG8f3OB",
      "answer": {
        "type": "noul",
        "noul": 0.38
      },
      "latency_s": 0.2445428330102004,
      "run": "main",
      "record_index": 90
    },
    {
      "case_id": "probability-020",
      "experiment": "probability",
      "target": 0.5458937198067633,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 91,
      "started_at": "2026-10-02T08:43:27.156220+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 1, \"B\": 2}, \"boxes\": {\"A\": {\"marked\": 14, \"unmarked\": 32}, \"B\": {\"marked\": 16, \"unmarked\": 8}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.4,
      "valid": true,
      "error": null,
      "cost_usd": 4.39e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 22,
        "total_tokens": 351,
        "cost": 4.39e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.39e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930607-AERSSBB0R9KHTfJxpTab",
        "object": "chat.completion",
        "created": 1790930607,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.4}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 22,
          "total_tokens": 351,
          "cost": 4.39e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.39e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930607-AERSSBB0R9KHTfJxpTab",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.4
      },
      "latency_s": 0.9724408339825459,
      "run": "main",
      "record_index": 91
    },
    {
      "case_id": "probability-040",
      "experiment": "probability",
      "target": 0.8317444219066937,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 92,
      "started_at": "2026-10-02T08:43:28.129132+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 1, \"B\": 9}, \"boxes\": {\"A\": {\"marked\": 19, \"unmarked\": 15}, \"B\": {\"marked\": 25, \"unmarked\": 4}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.547,
      "valid": true,
      "error": null,
      "cost_usd": 4.39e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 22,
        "total_tokens": 351,
        "cost": 4.39e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.39e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930608-S9sXizx1dT9tGZ7efSbP",
        "object": "chat.completion",
        "created": 1790930608,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.547}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 22,
          "total_tokens": 351,
          "cost": 4.39e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.39e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930608-S9sXizx1dT9tGZ7efSbP",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.547
      },
      "latency_s": 1.234619124996243,
      "run": "main",
      "record_index": 92
    },
    {
      "case_id": "probability-040",
      "experiment": "probability",
      "target": 0.8317444219066937,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 93,
      "started_at": "2026-10-02T08:43:29.364545+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 1,
            "B": 9
          },
          "boxes": {
            "A": {
              "marked": 19,
              "unmarked": 15
            },
            "B": {
              "marked": 25,
              "unmarked": 4
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.72,
      "valid": true,
      "error": null,
      "cost_usd": 2.0034e-05,
      "usage": {
        "input_tokens": 477,
        "output_tokens": 20,
        "cost": 2.0034e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.72
          }
        },
        "usage": {
          "input_tokens": 477,
          "output_tokens": 20,
          "cost": 2.0034e-05
        },
        "id": "gen-dec-1790930609-RzCeGocWmBD9e65p7oYf",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930609-RzCeGocWmBD9e65p7oYf",
      "answer": {
        "type": "noul",
        "noul": 0.72
      },
      "latency_s": 0.23903687499114312,
      "run": "main",
      "record_index": 93
    },
    {
      "case_id": "boolq-dev-3165",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 94,
      "started_at": "2026-10-02T08:43:29.603934+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is red stripe light sold in the us?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"In the U.S. in 2010, the bottling size was reduced from a typical 12 oz. per serving to 11.2 oz. per serving which is equivalent to the typical metric serving of 0.33L.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5,
      "valid": true,
      "error": null,
      "cost_usd": 3.64e-05,
      "usage": {
        "prompt_tokens": 254,
        "completion_tokens": 22,
        "total_tokens": 276,
        "cost": 3.64e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.64e-05,
          "upstream_inference_prompt_cost": 2.54e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930609-2rQ5ti4zONWiTEr5v2lh",
        "object": "chat.completion",
        "created": 1790930609,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 254,
          "completion_tokens": 22,
          "total_tokens": 276,
          "cost": 3.64e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.64e-05,
            "upstream_inference_prompt_cost": 2.54e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930609-2rQ5ti4zONWiTEr5v2lh",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5
      },
      "latency_s": 1.2559313340170775,
      "run": "main",
      "record_index": 94
    },
    {
      "case_id": "boolq-dev-3165",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 95,
      "started_at": "2026-10-02T08:43:30.860330+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "In the U.S. in 2010, the bottling size was reduced from a typical 12 oz. per serving to 11.2 oz. per serving which is equivalent to the typical metric serving of 0.33L."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is red stripe light sold in the us?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.63,
      "valid": true,
      "error": null,
      "cost_usd": 1.6044e-05,
      "usage": {
        "input_tokens": 382,
        "output_tokens": 20,
        "cost": 1.6044e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.63
          }
        },
        "usage": {
          "input_tokens": 382,
          "output_tokens": 20,
          "cost": 1.6044e-05
        },
        "id": "gen-dec-1790930610-y8YOEWkgKH7y5pwG7J69",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930610-y8YOEWkgKH7y5pwG7J69",
      "answer": {
        "type": "noul",
        "noul": 0.63
      },
      "latency_s": 0.29790974999195896,
      "run": "main",
      "record_index": 95
    },
    {
      "case_id": "probability-024",
      "experiment": "probability",
      "target": 0.6759183673469388,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 96,
      "started_at": "2026-10-02T08:43:31.158692+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 3, \"B\": 2}, \"boxes\": {\"A\": {\"marked\": 4, \"unmarked\": 1}, \"B\": {\"marked\": 24, \"unmarked\": 25}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.55,
      "valid": true,
      "error": null,
      "cost_usd": 4.39e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 22,
        "total_tokens": 351,
        "cost": 4.39e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.39e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930611-GdopSrqegZYruVs3mnt1",
        "object": "chat.completion",
        "created": 1790930611,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.55}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 22,
          "total_tokens": 351,
          "cost": 4.39e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.39e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930611-GdopSrqegZYruVs3mnt1",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.55
      },
      "latency_s": 2.749989749980159,
      "run": "main",
      "record_index": 96
    },
    {
      "case_id": "probability-024",
      "experiment": "probability",
      "target": 0.6759183673469388,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 97,
      "started_at": "2026-10-02T08:43:33.910245+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 3,
            "B": 2
          },
          "boxes": {
            "A": {
              "marked": 4,
              "unmarked": 1
            },
            "B": {
              "marked": 24,
              "unmarked": 25
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.5,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.5
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930613-xJrgD6jv6dZwsLtUvkqJ",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930613-xJrgD6jv6dZwsLtUvkqJ",
      "answer": {
        "type": "noul",
        "noul": 0.5
      },
      "latency_s": 0.24849929197807796,
      "run": "main",
      "record_index": 97
    },
    {
      "case_id": "boolq-dev-1171",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 98,
      "started_at": "2026-10-02T08:43:34.159179+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Relations between the European Union (EU) and Turkey were established in 1959 and the institutional framework is shaped formally since 1963 Ankara Agreement. Turkey is one of the EU's main partners in the Middle East and both are members of the European Union--Turkey Customs Union. The EU and Turkey have a common land border through the EU member states Bulgaria and Greece."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is turkey a part of the european union?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.26,
      "valid": true,
      "error": null,
      "cost_usd": 1.7052e-05,
      "usage": {
        "input_tokens": 406,
        "output_tokens": 20,
        "cost": 1.7052e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.26
          }
        },
        "usage": {
          "input_tokens": 406,
          "output_tokens": 20,
          "cost": 1.7052e-05
        },
        "id": "gen-dec-1790930614-KgRT8lWMCa51W5HENF3f",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930614-KgRT8lWMCa51W5HENF3f",
      "answer": {
        "type": "noul",
        "noul": 0.26
      },
      "latency_s": 0.22693729199818335,
      "run": "main",
      "record_index": 98
    },
    {
      "case_id": "boolq-dev-1171",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 99,
      "started_at": "2026-10-02T08:43:34.386559+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is turkey a part of the european union?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Relations between the European Union (EU) and Turkey were established in 1959 and the institutional framework is shaped formally since 1963 Ankara Agreement. Turkey is one of the EU's main partners in the Middle East and both are members of the European Union--Turkey Customs Union. The EU and Turkey have a common land border through the EU member states Bulgaria and Greece.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.79e-05,
      "usage": {
        "prompt_tokens": 279,
        "completion_tokens": 20,
        "total_tokens": 299,
        "cost": 3.79e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.79e-05,
          "upstream_inference_prompt_cost": 2.79e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930614-AcCadQvV28OjKD6xWHx4",
        "object": "chat.completion",
        "created": 1790930614,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 279,
          "completion_tokens": 20,
          "total_tokens": 299,
          "cost": 3.79e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.79e-05,
            "upstream_inference_prompt_cost": 2.79e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930614-AcCadQvV28OjKD6xWHx4",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.3310837919998448,
      "run": "main",
      "record_index": 99
    },
    {
      "case_id": "probability-028",
      "experiment": "probability",
      "target": 0.8259893920848633,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 100,
      "started_at": "2026-10-02T08:43:35.721544+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 1,
            "B": 5
          },
          "boxes": {
            "A": {
              "marked": 8,
              "unmarked": 11
            },
            "B": {
              "marked": 39,
              "unmarked": 4
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.68,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.68
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930615-PM9QleNBT5Xato3xhUdi",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930615-PM9QleNBT5Xato3xhUdi",
      "answer": {
        "type": "noul",
        "noul": 0.68
      },
      "latency_s": 0.22617150002042763,
      "run": "main",
      "record_index": 100
    },
    {
      "case_id": "probability-028",
      "experiment": "probability",
      "target": 0.8259893920848633,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 101,
      "started_at": "2026-10-02T08:43:35.948259+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 1, \"B\": 5}, \"boxes\": {\"A\": {\"marked\": 8, \"unmarked\": 11}, \"B\": {\"marked\": 39, \"unmarked\": 4}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.8372,
      "valid": true,
      "error": null,
      "cost_usd": 4.44e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 23,
        "total_tokens": 352,
        "cost": 4.44e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.44e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930615-nh5XQVULUR463EQLsE2l",
        "object": "chat.completion",
        "created": 1790930615,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.8372}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 23,
          "total_tokens": 352,
          "cost": 4.44e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.44e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930615-nh5XQVULUR463EQLsE2l",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.8372
      },
      "latency_s": 0.9990227910166141,
      "run": "main",
      "record_index": 101
    },
    {
      "case_id": "policy-027",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 102,
      "started_at": "2026-10-02T08:43:36.948865+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 0, \"defective\": true, \"final_sale\": true, \"price_credits\": 9, \"recalled\": false, \"receipt\": true, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930616-R7KJiOwp7avsfKIZalDG",
        "object": "chat.completion",
        "created": 1790930616,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930616-R7KJiOwp7avsfKIZalDG",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0625968750100583,
      "run": "main",
      "record_index": 102
    },
    {
      "case_id": "policy-027",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 103,
      "started_at": "2026-10-02T08:43:38.012121+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 0,
            "defective": true,
            "final_sale": true,
            "price_credits": 9,
            "recalled": false,
            "receipt": true,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 2.0286e-05,
      "usage": {
        "input_tokens": 483,
        "output_tokens": 20,
        "cost": 2.0286e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 483,
          "output_tokens": 20,
          "cost": 2.0286e-05
        },
        "id": "gen-dec-1790930618-16TA5U0sLRjxBhSRoHvB",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930618-16TA5U0sLRjxBhSRoHvB",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.2282087909989059,
      "run": "main",
      "record_index": 103
    },
    {
      "case_id": "policy-015",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 104,
      "started_at": "2026-10-02T08:43:38.241477+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 31,
            "defective": false,
            "final_sale": false,
            "price_credits": 247,
            "recalled": true,
            "receipt": true,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930618-6DqhbSE4WRJFWeQcLIud",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930618-6DqhbSE4WRJFWeQcLIud",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.24593737500254065,
      "run": "main",
      "record_index": 104
    },
    {
      "case_id": "policy-015",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 105,
      "started_at": "2026-10-02T08:43:38.488326+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 31, \"defective\": false, \"final_sale\": false, \"price_credits\": 247, \"recalled\": true, \"receipt\": true, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930618-26XKvhfqZafpKKobHpU2",
        "object": "chat.completion",
        "created": 1790930618,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930618-26XKvhfqZafpKKobHpU2",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0160732500080485,
      "run": "main",
      "record_index": 105
    },
    {
      "case_id": "probability-017",
      "experiment": "probability",
      "target": 0.5166666666666667,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 106,
      "started_at": "2026-10-02T08:43:39.504880+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 7, \"unmarked\": 10}, \"B\": {\"marked\": 31, \"unmarked\": 29}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5166666667,
      "valid": true,
      "error": null,
      "cost_usd": 4.16e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 25,
        "total_tokens": 316,
        "cost": 4.16e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.16e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930619-95YfARB5WunYj49W0dKB",
        "object": "chat.completion",
        "created": 1790930619,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5166666667}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 25,
          "total_tokens": 316,
          "cost": 4.16e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.16e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930619-95YfARB5WunYj49W0dKB",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5166666667
      },
      "latency_s": 1.451190542022232,
      "run": "main",
      "record_index": 106
    },
    {
      "case_id": "probability-017",
      "experiment": "probability",
      "target": 0.5166666666666667,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 107,
      "started_at": "2026-10-02T08:43:40.956566+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 7,
              "unmarked": 10
            },
            "B": {
              "marked": 31,
              "unmarked": 29
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.56,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.56
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930620-8ahybHfZB2wRiAd1dQ5q",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930620-8ahybHfZB2wRiAd1dQ5q",
      "answer": {
        "type": "noul",
        "noul": 0.56
      },
      "latency_s": 0.220227874990087,
      "run": "main",
      "record_index": 107
    },
    {
      "case_id": "boolq-dev-2552",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 108,
      "started_at": "2026-10-02T08:43:41.177179+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Lily of the valley (Convallaria majalis /ˌkɒnvəˈleɪriə məˈdʒeɪlɪs/), sometimes written lily-of-the-valley, is a sweetly scented, highly poisonous woodland flowering plant that is native throughout the cool temperate Northern Hemisphere in Asia and Europe. Other names include May bells, Our Lady's tears, and Mary's tears. Its French name, muguet, sometimes appears in the names of perfumes imitating the flower's scent."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is lily-of-the-valley poisonous?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 1.89e-05,
      "usage": {
        "input_tokens": 450,
        "output_tokens": 20,
        "cost": 1.89e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 450,
          "output_tokens": 20,
          "cost": 1.89e-05
        },
        "id": "gen-dec-1790930621-cbssFEoISwIC4gKoS5YX",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930621-cbssFEoISwIC4gKoS5YX",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.25915129101485945,
      "run": "main",
      "record_index": 108
    },
    {
      "case_id": "boolq-dev-2552",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 109,
      "started_at": "2026-10-02T08:43:41.436753+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is lily-of-the-valley poisonous?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Lily of the valley (Convallaria majalis /ˌkɒnvəˈleɪriə məˈdʒeɪlɪs/), sometimes written lily-of-the-valley, is a sweetly scented, highly poisonous woodland flowering plant that is native throughout the cool temperate Northern Hemisphere in Asia and Europe. Other names include May bells, Our Lady's tears, and Mary's tears. Its French name, muguet, sometimes appears in the names of perfumes imitating the flower's scent.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.16e-05,
      "usage": {
        "prompt_tokens": 316,
        "completion_tokens": 20,
        "total_tokens": 336,
        "cost": 4.16e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.16e-05,
          "upstream_inference_prompt_cost": 3.16e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930621-9ML8y6dP0bhrlrPVTZwl",
        "object": "chat.completion",
        "created": 1790930621,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 316,
          "completion_tokens": 20,
          "total_tokens": 336,
          "cost": 4.16e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.16e-05,
            "upstream_inference_prompt_cost": 3.16e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930621-9ML8y6dP0bhrlrPVTZwl",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1404950830037706,
      "run": "main",
      "record_index": 109
    },
    {
      "case_id": "probability-052",
      "experiment": "probability",
      "target": 0.483974358974359,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 110,
      "started_at": "2026-10-02T08:43:42.577728+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 7, \"B\": 9}, \"boxes\": {\"A\": {\"marked\": 3, \"unmarked\": 15}, \"B\": {\"marked\": 19, \"unmarked\": 7}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5247252747,
      "valid": true,
      "error": null,
      "cost_usd": 4.54e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 25,
        "total_tokens": 354,
        "cost": 4.54e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.54e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930622-LvwP9DSNZt5wt8VbIyCW",
        "object": "chat.completion",
        "created": 1790930622,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5247252747}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 25,
          "total_tokens": 354,
          "cost": 4.54e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.54e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930622-LvwP9DSNZt5wt8VbIyCW",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5247252747
      },
      "latency_s": 1.0520879160030745,
      "run": "main",
      "record_index": 110
    },
    {
      "case_id": "probability-052",
      "experiment": "probability",
      "target": 0.483974358974359,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 111,
      "started_at": "2026-10-02T08:43:43.631304+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 7,
            "B": 9
          },
          "boxes": {
            "A": {
              "marked": 3,
              "unmarked": 15
            },
            "B": {
              "marked": 19,
              "unmarked": 7
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.55,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.55
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930623-gT2T4hqLH86HDnq6VMdz",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930623-gT2T4hqLH86HDnq6VMdz",
      "answer": {
        "type": "noul",
        "noul": 0.55
      },
      "latency_s": 0.23642499998095445,
      "run": "main",
      "record_index": 111
    },
    {
      "case_id": "probability-056",
      "experiment": "probability",
      "target": 0.7537037037037037,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 112,
      "started_at": "2026-10-02T08:43:43.869366+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 8,
            "B": 4
          },
          "boxes": {
            "A": {
              "marked": 21,
              "unmarked": 9
            },
            "B": {
              "marked": 31,
              "unmarked": 5
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.77,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.77
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930623-lS9FaRd8Q7okAnyfPeDW",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930623-lS9FaRd8Q7okAnyfPeDW",
      "answer": {
        "type": "noul",
        "noul": 0.77
      },
      "latency_s": 0.2210766249918379,
      "run": "main",
      "record_index": 112
    },
    {
      "case_id": "probability-056",
      "experiment": "probability",
      "target": 0.7537037037037037,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 113,
      "started_at": "2026-10-02T08:43:44.090892+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 8, \"B\": 4}, \"boxes\": {\"A\": {\"marked\": 21, \"unmarked\": 9}, \"B\": {\"marked\": 31, \"unmarked\": 5}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.7283,
      "valid": true,
      "error": null,
      "cost_usd": 4.44e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 23,
        "total_tokens": 352,
        "cost": 4.44e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.44e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930624-X65NjSAyI8fIQFiKTYzj",
        "object": "chat.completion",
        "created": 1790930624,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.7283}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 23,
          "total_tokens": 352,
          "cost": 4.44e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.44e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930624-X65NjSAyI8fIQFiKTYzj",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.7283
      },
      "latency_s": 1.879591667006025,
      "run": "main",
      "record_index": 113
    },
    {
      "case_id": "probability-032",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 114,
      "started_at": "2026-10-02T08:43:45.970928+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 4,
              "unmarked": 16
            },
            "B": {
              "marked": 11,
              "unmarked": 2
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "A",
          "selection": "Choose one item uniformly at random from group A only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.22,
      "valid": true,
      "error": null,
      "cost_usd": 1.8396e-05,
      "usage": {
        "input_tokens": 438,
        "output_tokens": 20,
        "cost": 1.8396e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.22
          }
        },
        "usage": {
          "input_tokens": 438,
          "output_tokens": 20,
          "cost": 1.8396e-05
        },
        "id": "gen-dec-1790930625-hCMd3En50MjvKfHpmu4x",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930625-hCMd3En50MjvKfHpmu4x",
      "answer": {
        "type": "noul",
        "noul": 0.22
      },
      "latency_s": 0.23355991698917933,
      "run": "main",
      "record_index": 114
    },
    {
      "case_id": "probability-032",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 115,
      "started_at": "2026-10-02T08:43:46.205694+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 4, \"unmarked\": 16}, \"B\": {\"marked\": 11, \"unmarked\": 2}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"A\", \"selection\": \"Choose one item uniformly at random from group A only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.2,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930626-eK7QpbjbkrhX4kJVxHf0",
        "object": "chat.completion",
        "created": 1790930626,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.2}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930626-eK7QpbjbkrhX4kJVxHf0",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.2
      },
      "latency_s": 0.9389851670130156,
      "run": "main",
      "record_index": 115
    },
    {
      "case_id": "boolq-dev-3125",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 116,
      "started_at": "2026-10-02T08:43:47.145157+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The circumzenithal arc, also called the circumzenith arc (CZA), upside-down rainbow, and the Bravais arc, is an optical phenomenon similar in appearance to a rainbow, but belonging to the family of halos arising from refraction of sunlight through ice crystals, generally in cirrus or cirrostratus clouds, rather than from raindrops. The arc is located at a considerable distance (approximately 46°) above the sun and at most forms a quarter of a circle centered on the zenith. It has been called ``a smile in the sky'', its first impression being that of an upside-down rainbow. The CZA is one of the brightest and most colorful members of the halo family. Its colors, ranging from violet on top to red at the bottom, are purer than those of a rainbow because there is much less overlap in their formation."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is there such a thing as an upside down rainbow?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.97,
      "valid": true,
      "error": null,
      "cost_usd": 2.1546e-05,
      "usage": {
        "input_tokens": 513,
        "output_tokens": 20,
        "cost": 2.1546e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.97
          }
        },
        "usage": {
          "input_tokens": 513,
          "output_tokens": 20,
          "cost": 2.1546e-05
        },
        "id": "gen-dec-1790930627-lUagKvv6Q2qK4T6lHXse",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930627-lUagKvv6Q2qK4T6lHXse",
      "answer": {
        "type": "noul",
        "noul": 0.97
      },
      "latency_s": 0.215893791988492,
      "run": "main",
      "record_index": 116
    },
    {
      "case_id": "boolq-dev-3125",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 117,
      "started_at": "2026-10-02T08:43:47.361452+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is there such a thing as an upside down rainbow?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The circumzenithal arc, also called the circumzenith arc (CZA), upside-down rainbow, and the Bravais arc, is an optical phenomenon similar in appearance to a rainbow, but belonging to the family of halos arising from refraction of sunlight through ice crystals, generally in cirrus or cirrostratus clouds, rather than from raindrops. The arc is located at a considerable distance (approximately 46°) above the sun and at most forms a quarter of a circle centered on the zenith. It has been called ``a smile in the sky'', its first impression being that of an upside-down rainbow. The CZA is one of the brightest and most colorful members of the halo family. Its colors, ranging from violet on top to red at the bottom, are purer than those of a rainbow because there is much less overlap in their formation.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.84e-05,
      "usage": {
        "prompt_tokens": 384,
        "completion_tokens": 20,
        "total_tokens": 404,
        "cost": 4.84e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.84e-05,
          "upstream_inference_prompt_cost": 3.84e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930627-Pi192B4oXPkImOhkNSFH",
        "object": "chat.completion",
        "created": 1790930627,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 384,
          "completion_tokens": 20,
          "total_tokens": 404,
          "cost": 4.84e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.84e-05,
            "upstream_inference_prompt_cost": 3.84e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930627-Pi192B4oXPkImOhkNSFH",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.2944159170147032,
      "run": "main",
      "record_index": 117
    },
    {
      "case_id": "boolq-dev-2098",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 118,
      "started_at": "2026-10-02T08:43:48.656336+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is a brother in law considered a relative?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The immediate family is a defined group of relations, used in rules or laws to determine which members of a person's family are affected by those rules. It normally includes a person's parents, spouses, siblings, children, or an individual related by blood whose close association is an equivalent of a family relationship. It can contain others connected by birth, adoption, marriage, civil partnership, or cohabitation, such as grandparents, great-grandparents, grandchildren, great-grandchildren, aunts, uncles, siblings-in-law, half-siblings, cousins, adopted children and step-parents/step-children, and cohabiting partners.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.31e-05,
      "usage": {
        "prompt_tokens": 331,
        "completion_tokens": 20,
        "total_tokens": 351,
        "cost": 4.31e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.31e-05,
          "upstream_inference_prompt_cost": 3.31e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930628-QJoXwMXsRYl2aLtmFSC2",
        "object": "chat.completion",
        "created": 1790930628,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 331,
          "completion_tokens": 20,
          "total_tokens": 351,
          "cost": 4.31e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.31e-05,
            "upstream_inference_prompt_cost": 3.31e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930628-QJoXwMXsRYl2aLtmFSC2",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.5007001250050962,
      "run": "main",
      "record_index": 118
    },
    {
      "case_id": "boolq-dev-2098",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 119,
      "started_at": "2026-10-02T08:43:50.158209+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The immediate family is a defined group of relations, used in rules or laws to determine which members of a person's family are affected by those rules. It normally includes a person's parents, spouses, siblings, children, or an individual related by blood whose close association is an equivalent of a family relationship. It can contain others connected by birth, adoption, marriage, civil partnership, or cohabitation, such as grandparents, great-grandparents, grandchildren, great-grandchildren, aunts, uncles, siblings-in-law, half-siblings, cousins, adopted children and step-parents/step-children, and cohabiting partners."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is a brother in law considered a relative?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.95,
      "valid": true,
      "error": null,
      "cost_usd": 1.9404e-05,
      "usage": {
        "input_tokens": 462,
        "output_tokens": 20,
        "cost": 1.9404e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.95
          }
        },
        "usage": {
          "input_tokens": 462,
          "output_tokens": 20,
          "cost": 1.9404e-05
        },
        "id": "gen-dec-1790930630-Uo2jtB8MTFFvGJ6ms9Qt",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930630-Uo2jtB8MTFFvGJ6ms9Qt",
      "answer": {
        "type": "noul",
        "noul": 0.95
      },
      "latency_s": 0.22809937500278465,
      "run": "main",
      "record_index": 119
    },
    {
      "case_id": "policy-046",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 120,
      "started_at": "2026-10-02T08:43:50.386809+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": false, \"days_since_purchase\": 29, \"defective\": false, \"final_sale\": false, \"price_credits\": 208, \"recalled\": false, \"receipt\": false, \"unused\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930630-D3iTg3KaUJkyMzPFEI0E",
        "object": "chat.completion",
        "created": 1790930630,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930630-D3iTg3KaUJkyMzPFEI0E",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.1114752920111641,
      "run": "main",
      "record_index": 120
    },
    {
      "case_id": "policy-046",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 121,
      "started_at": "2026-10-02T08:43:51.499200+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": false,
            "days_since_purchase": 29,
            "defective": false,
            "final_sale": false,
            "price_credits": 208,
            "recalled": false,
            "receipt": false,
            "unused": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.07,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.07
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930631-OTunErrI01clN2uSscW9",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930631-OTunErrI01clN2uSscW9",
      "answer": {
        "type": "noul",
        "noul": 0.07
      },
      "latency_s": 0.26357270902371965,
      "run": "main",
      "record_index": 121
    },
    {
      "case_id": "probability-048",
      "experiment": "probability",
      "target": 0.07777777777777778,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 122,
      "started_at": "2026-10-02T08:43:51.763251+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 8, \"B\": 7}, \"boxes\": {\"A\": {\"marked\": 7, \"unmarked\": 41}, \"B\": {\"marked\": 0, \"unmarked\": 8}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0777777778,
      "valid": true,
      "error": null,
      "cost_usd": 4.54e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 25,
        "total_tokens": 354,
        "cost": 4.54e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.54e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930631-of02swOwm1bWweLGvSrz",
        "object": "chat.completion",
        "created": 1790930631,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.0777777778}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 25,
          "total_tokens": 354,
          "cost": 4.54e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.54e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930631-of02swOwm1bWweLGvSrz",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.0777777778
      },
      "latency_s": 0.9450222910090815,
      "run": "main",
      "record_index": 122
    },
    {
      "case_id": "probability-048",
      "experiment": "probability",
      "target": 0.07777777777777778,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 123,
      "started_at": "2026-10-02T08:43:52.709521+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 8,
            "B": 7
          },
          "boxes": {
            "A": {
              "marked": 7,
              "unmarked": 41
            },
            "B": {
              "marked": 0,
              "unmarked": 8
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.17,
      "valid": true,
      "error": null,
      "cost_usd": 1.995e-05,
      "usage": {
        "input_tokens": 475,
        "output_tokens": 20,
        "cost": 1.995e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.17
          }
        },
        "usage": {
          "input_tokens": 475,
          "output_tokens": 20,
          "cost": 1.995e-05
        },
        "id": "gen-dec-1790930632-mFmoKuarDLCc0risFVl7",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930632-mFmoKuarDLCc0risFVl7",
      "answer": {
        "type": "noul",
        "noul": 0.17
      },
      "latency_s": 0.3222276669985149,
      "run": "main",
      "record_index": 123
    },
    {
      "case_id": "policy-002",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 124,
      "started_at": "2026-10-02T08:43:53.032571+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "green",
            "clearance": 2,
            "emergency_permit": true,
            "experience_years": 15,
            "role": "staff",
            "sponsor": false,
            "suspended": false,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930633-6dc9WvxLSbeI7YxYsdtD",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930633-6dc9WvxLSbeI7YxYsdtD",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.22812149999663234,
      "run": "main",
      "record_index": 124
    },
    {
      "case_id": "policy-002",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 125,
      "started_at": "2026-10-02T08:43:53.261087+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"green\", \"clearance\": 2, \"emergency_permit\": true, \"experience_years\": 15, \"role\": \"staff\", \"sponsor\": false, \"suspended\": false, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930633-XEmgXmJFmVx9gZpFocF0",
        "object": "chat.completion",
        "created": 1790930633,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930633-XEmgXmJFmVx9gZpFocF0",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.478356999985408,
      "run": "main",
      "record_index": 125
    },
    {
      "case_id": "policy-056",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 126,
      "started_at": "2026-10-02T08:43:54.740064+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "green",
            "clearance": 3,
            "emergency_permit": false,
            "experience_years": 8,
            "role": "contractor",
            "sponsor": true,
            "suspended": false,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930634-YhpPEnGTrT81FhZ4BtWU",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930634-YhpPEnGTrT81FhZ4BtWU",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.2424692919885274,
      "run": "main",
      "record_index": 126
    },
    {
      "case_id": "policy-056",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 127,
      "started_at": "2026-10-02T08:43:54.982970+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"green\", \"clearance\": 3, \"emergency_permit\": false, \"experience_years\": 8, \"role\": \"contractor\", \"sponsor\": true, \"suspended\": false, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.62e-05,
      "usage": {
        "prompt_tokens": 362,
        "completion_tokens": 20,
        "total_tokens": 382,
        "cost": 4.62e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.62e-05,
          "upstream_inference_prompt_cost": 3.62e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930635-AXl7RgFzntWiJWJLtN56",
        "object": "chat.completion",
        "created": 1790930635,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 362,
          "completion_tokens": 20,
          "total_tokens": 382,
          "cost": 4.62e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.62e-05,
            "upstream_inference_prompt_cost": 3.62e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930635-AXl7RgFzntWiJWJLtN56",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9691010419919621,
      "run": "main",
      "record_index": 127
    },
    {
      "case_id": "policy-043",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 128,
      "started_at": "2026-10-02T08:43:55.953167+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 31, \"defective\": true, \"final_sale\": true, \"price_credits\": 214, \"recalled\": true, \"receipt\": true, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930635-Y4oHgcn6TGu30ZL1lmZD",
        "object": "chat.completion",
        "created": 1790930635,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930635-Y4oHgcn6TGu30ZL1lmZD",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.4186420000041835,
      "run": "main",
      "record_index": 128
    },
    {
      "case_id": "policy-043",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 129,
      "started_at": "2026-10-02T08:43:57.372316+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 31,
            "defective": true,
            "final_sale": true,
            "price_credits": 214,
            "recalled": true,
            "receipt": true,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930637-lcgXQM0mW7J84LDk0CVk",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930637-lcgXQM0mW7J84LDk0CVk",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.33430012498865835,
      "run": "main",
      "record_index": 129
    },
    {
      "case_id": "probability-005",
      "experiment": "probability",
      "target": 0.025,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 130,
      "started_at": "2026-10-02T08:43:57.708393+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 1,
              "unmarked": 39
            },
            "B": {
              "marked": 7,
              "unmarked": 26
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "A",
          "selection": "Choose one item uniformly at random from group A only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.06,
      "valid": true,
      "error": null,
      "cost_usd": 1.8396e-05,
      "usage": {
        "input_tokens": 438,
        "output_tokens": 20,
        "cost": 1.8396e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.06
          }
        },
        "usage": {
          "input_tokens": 438,
          "output_tokens": 20,
          "cost": 1.8396e-05
        },
        "id": "gen-dec-1790930637-ybGQe0KSSDttc90liRnU",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930637-ybGQe0KSSDttc90liRnU",
      "answer": {
        "type": "noul",
        "noul": 0.06
      },
      "latency_s": 0.24782554199919105,
      "run": "main",
      "record_index": 130
    },
    {
      "case_id": "probability-005",
      "experiment": "probability",
      "target": 0.025,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 131,
      "started_at": "2026-10-02T08:43:57.956723+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 1, \"unmarked\": 39}, \"B\": {\"marked\": 7, \"unmarked\": 26}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"A\", \"selection\": \"Choose one item uniformly at random from group A only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.025,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930637-ikGNiOZRzQmPo7qvITr2",
        "object": "chat.completion",
        "created": 1790930637,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.025}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930637-ikGNiOZRzQmPo7qvITr2",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.025
      },
      "latency_s": 1.3008887089963537,
      "run": "main",
      "record_index": 131
    },
    {
      "case_id": "probability-021",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 132,
      "started_at": "2026-10-02T08:43:59.258111+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 13,
              "unmarked": 64
            },
            "B": {
              "marked": 10,
              "unmarked": 40
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.22,
      "valid": true,
      "error": null,
      "cost_usd": 1.848e-05,
      "usage": {
        "input_tokens": 440,
        "output_tokens": 20,
        "cost": 1.848e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.22
          }
        },
        "usage": {
          "input_tokens": 440,
          "output_tokens": 20,
          "cost": 1.848e-05
        },
        "id": "gen-dec-1790930639-ZBu5jjxXviFA78o7qjES",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930639-ZBu5jjxXviFA78o7qjES",
      "answer": {
        "type": "noul",
        "noul": 0.22
      },
      "latency_s": 0.2541274999966845,
      "run": "main",
      "record_index": 132
    },
    {
      "case_id": "probability-021",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 133,
      "started_at": "2026-10-02T08:43:59.512661+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 13, \"unmarked\": 64}, \"B\": {\"marked\": 10, \"unmarked\": 40}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.2,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930639-F0Zbbvlku0lZdevW5mHc",
        "object": "chat.completion",
        "created": 1790930639,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.2}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930639-F0Zbbvlku0lZdevW5mHc",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.2
      },
      "latency_s": 1.3724771669949405,
      "run": "main",
      "record_index": 133
    },
    {
      "case_id": "boolq-dev-1517",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 134,
      "started_at": "2026-10-02T08:44:00.885627+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The 2018 FIFA World Cup was the 21st FIFA World Cup, an international football tournament contested by the men's national teams of the member associations of FIFA once every four years. It took place in Russia from 14 June to 15 July 2018. It was the first World Cup to be held in Eastern Europe, and the 11th time that it had been held in Europe. At an estimated cost of over $14.2 billion, it was the most expensive World Cup. It was also the first World Cup to use the video assistant referee (VAR) system."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are all world cup matches played in russia?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.86,
      "valid": true,
      "error": null,
      "cost_usd": 1.9236e-05,
      "usage": {
        "input_tokens": 458,
        "output_tokens": 20,
        "cost": 1.9236e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.86
          }
        },
        "usage": {
          "input_tokens": 458,
          "output_tokens": 20,
          "cost": 1.9236e-05
        },
        "id": "gen-dec-1790930640-LWAvWJA9sGOmSGhEI1Be",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930640-LWAvWJA9sGOmSGhEI1Be",
      "answer": {
        "type": "noul",
        "noul": 0.86
      },
      "latency_s": 0.23572833399521187,
      "run": "main",
      "record_index": 134
    },
    {
      "case_id": "boolq-dev-1517",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 135,
      "started_at": "2026-10-02T08:44:01.121897+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are all world cup matches played in russia?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The 2018 FIFA World Cup was the 21st FIFA World Cup, an international football tournament contested by the men's national teams of the member associations of FIFA once every four years. It took place in Russia from 14 June to 15 July 2018. It was the first World Cup to be held in Eastern Europe, and the 11th time that it had been held in Europe. At an estimated cost of over $14.2 billion, it was the most expensive World Cup. It was also the first World Cup to use the video assistant referee (VAR) system.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.25e-05,
      "usage": {
        "prompt_tokens": 325,
        "completion_tokens": 20,
        "total_tokens": 345,
        "cost": 4.25e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.25e-05,
          "upstream_inference_prompt_cost": 3.25e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930641-fyobihYk8PuvPetyqVRy",
        "object": "chat.completion",
        "created": 1790930641,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 325,
          "completion_tokens": 20,
          "total_tokens": 345,
          "cost": 4.25e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.25e-05,
            "upstream_inference_prompt_cost": 3.25e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930641-fyobihYk8PuvPetyqVRy",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.2935594159935135,
      "run": "main",
      "record_index": 135
    },
    {
      "case_id": "probability-023",
      "experiment": "probability",
      "target": 0.1978021978021978,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 136,
      "started_at": "2026-10-02T08:44:02.415916+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 4,
            "B": 9
          },
          "boxes": {
            "A": {
              "marked": 18,
              "unmarked": 10
            },
            "B": {
              "marked": 0,
              "unmarked": 5
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.58,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.58
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930642-haaG8I5UMCs9OIlV3KCU",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930642-haaG8I5UMCs9OIlV3KCU",
      "answer": {
        "type": "noul",
        "noul": 0.58
      },
      "latency_s": 0.22387637497740798,
      "run": "main",
      "record_index": 136
    },
    {
      "case_id": "probability-023",
      "experiment": "probability",
      "target": 0.1978021978021978,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 137,
      "started_at": "2026-10-02T08:44:02.641330+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 4, \"B\": 9}, \"boxes\": {\"A\": {\"marked\": 18, \"unmarked\": 10}, \"B\": {\"marked\": 0, \"unmarked\": 5}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.36,
      "valid": true,
      "error": null,
      "cost_usd": 4.39e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 22,
        "total_tokens": 351,
        "cost": 4.39e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.39e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930642-Tt5v0r52wGZO98E4YWKr",
        "object": "chat.completion",
        "created": 1790930642,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.36}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 22,
          "total_tokens": 351,
          "cost": 4.39e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.39e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930642-Tt5v0r52wGZO98E4YWKr",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.36
      },
      "latency_s": 1.0144949579844251,
      "run": "main",
      "record_index": 137
    },
    {
      "case_id": "probability-022",
      "experiment": "probability",
      "target": 0.6976495726495726,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 138,
      "started_at": "2026-10-02T08:44:03.656350+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 3, \"B\": 3}, \"boxes\": {\"A\": {\"marked\": 17, \"unmarked\": 19}, \"B\": {\"marked\": 12, \"unmarked\": 1}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.4446,
      "valid": true,
      "error": null,
      "cost_usd": 4.44e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 23,
        "total_tokens": 352,
        "cost": 4.44e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.44e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930643-plb8a4HMiCJJfAQSJF3N",
        "object": "chat.completion",
        "created": 1790930643,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.4446}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 23,
          "total_tokens": 352,
          "cost": 4.44e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.44e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930643-plb8a4HMiCJJfAQSJF3N",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.4446
      },
      "latency_s": 1.0172259579994716,
      "run": "main",
      "record_index": 138
    },
    {
      "case_id": "probability-022",
      "experiment": "probability",
      "target": 0.6976495726495726,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 139,
      "started_at": "2026-10-02T08:44:04.674649+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 3,
            "B": 3
          },
          "boxes": {
            "A": {
              "marked": 17,
              "unmarked": 19
            },
            "B": {
              "marked": 12,
              "unmarked": 1
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.64,
      "valid": true,
      "error": null,
      "cost_usd": 2.0034e-05,
      "usage": {
        "input_tokens": 477,
        "output_tokens": 20,
        "cost": 2.0034e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.64
          }
        },
        "usage": {
          "input_tokens": 477,
          "output_tokens": 20,
          "cost": 2.0034e-05
        },
        "id": "gen-dec-1790930644-jiCqYDo0k0qORnmP6KzX",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930644-jiCqYDo0k0qORnmP6KzX",
      "answer": {
        "type": "noul",
        "noul": 0.64
      },
      "latency_s": 0.2365575839939993,
      "run": "main",
      "record_index": 139
    },
    {
      "case_id": "boolq-dev-1859",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 140,
      "started_at": "2026-10-02T08:44:04.911554+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Pinot gris, pinot grigio or Grauburgunder is a white wine grape variety of the species Vitis vinifera. Thought to be a mutant clone of the pinot noir variety, it normally has a grayish-blue fruit, accounting for its name but the grapes can have a brownish pink to black and even white appearance. The word pinot could have been given to it because the grapes grow in small pine cone-shaped clusters. The wines produced from this grape also vary in color from a deep golden yellow to copper and even a light shade of pink, and it is one of the more popular grapes for skin-contact wine."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are pinot noir and pinot gris the same grape?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.17,
      "valid": true,
      "error": null,
      "cost_usd": 1.9446e-05,
      "usage": {
        "input_tokens": 463,
        "output_tokens": 20,
        "cost": 1.9446e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.17
          }
        },
        "usage": {
          "input_tokens": 463,
          "output_tokens": 20,
          "cost": 1.9446e-05
        },
        "id": "gen-dec-1790930644-9JdjtqNXJXG2qmQxQ48G",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930644-9JdjtqNXJXG2qmQxQ48G",
      "answer": {
        "type": "noul",
        "noul": 0.17
      },
      "latency_s": 0.23655870798393153,
      "run": "main",
      "record_index": 140
    },
    {
      "case_id": "boolq-dev-1859",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 141,
      "started_at": "2026-10-02T08:44:05.148541+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are pinot noir and pinot gris the same grape?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Pinot gris, pinot grigio or Grauburgunder is a white wine grape variety of the species Vitis vinifera. Thought to be a mutant clone of the pinot noir variety, it normally has a grayish-blue fruit, accounting for its name but the grapes can have a brownish pink to black and even white appearance. The word pinot could have been given to it because the grapes grow in small pine cone-shaped clusters. The wines produced from this grape also vary in color from a deep golden yellow to copper and even a light shade of pink, and it is one of the more popular grapes for skin-contact wine.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.05,
      "valid": true,
      "error": null,
      "cost_usd": 4.49e-05,
      "usage": {
        "prompt_tokens": 339,
        "completion_tokens": 22,
        "total_tokens": 361,
        "cost": 4.49e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.49e-05,
          "upstream_inference_prompt_cost": 3.39e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930645-1sejo6LFXRGQ0lvuaFQn",
        "object": "chat.completion",
        "created": 1790930645,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.05}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 339,
          "completion_tokens": 22,
          "total_tokens": 361,
          "cost": 4.49e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.49e-05,
            "upstream_inference_prompt_cost": 3.39e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930645-1sejo6LFXRGQ0lvuaFQn",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.05
      },
      "latency_s": 1.2449191670166329,
      "run": "main",
      "record_index": 141
    },
    {
      "case_id": "probability-027",
      "experiment": "probability",
      "target": 0.23333333333333334,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 142,
      "started_at": "2026-10-02T08:44:06.394326+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is exactly one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 2, \"unmarked\": 14}, \"event\": \"exactly_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.23333333333333334,
      "valid": true,
      "error": null,
      "cost_usd": 4.15e-05,
      "usage": {
        "prompt_tokens": 280,
        "completion_tokens": 27,
        "total_tokens": 307,
        "cost": 4.15e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.15e-05,
          "upstream_inference_prompt_cost": 2.8e-05,
          "upstream_inference_completions_cost": 1.35e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930646-x5RPXm2BFF4TaD9wwm6E",
        "object": "chat.completion",
        "created": 1790930646,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.23333333333333334}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 280,
          "completion_tokens": 27,
          "total_tokens": 307,
          "cost": 4.15e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.15e-05,
            "upstream_inference_prompt_cost": 2.8e-05,
            "upstream_inference_completions_cost": 1.35e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930646-x5RPXm2BFF4TaD9wwm6E",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.23333333333333334
      },
      "latency_s": 1.101665124995634,
      "run": "main",
      "record_index": 142
    },
    {
      "case_id": "probability-027",
      "experiment": "probability",
      "target": 0.23333333333333334,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 143,
      "started_at": "2026-10-02T08:44:07.496449+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 2,
            "unmarked": 14
          },
          "event": "exactly_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is exactly one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.36,
      "valid": true,
      "error": null,
      "cost_usd": 1.7598e-05,
      "usage": {
        "input_tokens": 419,
        "output_tokens": 20,
        "cost": 1.7598e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.36
          }
        },
        "usage": {
          "input_tokens": 419,
          "output_tokens": 20,
          "cost": 1.7598e-05
        },
        "id": "gen-dec-1790930647-eVXSmfyafEOk2zEFIL09",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930647-eVXSmfyafEOk2zEFIL09",
      "answer": {
        "type": "noul",
        "noul": 0.36
      },
      "latency_s": 0.23737462499411777,
      "run": "main",
      "record_index": 143
    },
    {
      "case_id": "policy-049",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 144,
      "started_at": "2026-10-02T08:44:07.734234+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 0,
            "defective": true,
            "final_sale": false,
            "price_credits": 222,
            "recalled": true,
            "receipt": true,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 2.037e-05,
      "usage": {
        "input_tokens": 485,
        "output_tokens": 20,
        "cost": 2.037e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 485,
          "output_tokens": 20,
          "cost": 2.037e-05
        },
        "id": "gen-dec-1790930647-iU0OCp7PBOJXYl8gLrLa",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930647-iU0OCp7PBOJXYl8gLrLa",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.26238866700441577,
      "run": "main",
      "record_index": 144
    },
    {
      "case_id": "policy-049",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 145,
      "started_at": "2026-10-02T08:44:07.997107+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 0, \"defective\": true, \"final_sale\": false, \"price_credits\": 222, \"recalled\": true, \"receipt\": true, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930648-OvdwIC609bg30jxnhNtO",
        "object": "chat.completion",
        "created": 1790930648,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930648-OvdwIC609bg30jxnhNtO",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9533589160128031,
      "run": "main",
      "record_index": 145
    },
    {
      "case_id": "policy-034",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 146,
      "started_at": "2026-10-02T08:44:08.952026+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "green",
            "clearance": 1,
            "emergency_permit": true,
            "experience_years": 0,
            "role": "visitor",
            "sponsor": false,
            "suspended": false,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1042e-05,
      "usage": {
        "input_tokens": 501,
        "output_tokens": 20,
        "cost": 2.1042e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 501,
          "output_tokens": 20,
          "cost": 2.1042e-05
        },
        "id": "gen-dec-1790930648-6BL88Kp4iJeikZTPg51N",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930648-6BL88Kp4iJeikZTPg51N",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.22284262502216734,
      "run": "main",
      "record_index": 146
    },
    {
      "case_id": "policy-034",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 147,
      "started_at": "2026-10-02T08:44:09.175284+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"green\", \"clearance\": 1, \"emergency_permit\": true, \"experience_years\": 0, \"role\": \"visitor\", \"sponsor\": false, \"suspended\": false, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930649-ftBlHOmiDzRuzEXT83fb",
        "object": "chat.completion",
        "created": 1790930649,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930649-ftBlHOmiDzRuzEXT83fb",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0147319170064293,
      "run": "main",
      "record_index": 147
    },
    {
      "case_id": "boolq-dev-1335",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 148,
      "started_at": "2026-10-02T08:44:10.191276+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are there nerves in the central nervous system?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The central nervous system (CNS) is the part of the nervous system consisting of the brain and spinal cord. The central nervous system is so named because it integrates information it receives from, and coordinates and influences the activity of, all parts of the bodies of bilaterally symmetric animals--that is, all multicellular animals except sponges and radially symmetric animals such as jellyfish--and it contains the majority of the nervous system. Many consider the retina and the optic nerve (cranial nerve II), as well as the olfactory nerves (cranial nerve I) and olfactory epithelium as parts of the CNS, synapsing directly on brain tissue without intermediate ganglia. As such, the olfactory epithelium is the only central nervous tissue in direct contact with the environment, which opens up for therapeutic treatments. The CNS is contained within the dorsal body cavity, with the brain housed in the cranial cavity and the spinal cord in the spinal canal. In vertebrates, the brain is protected by the skull, while the spinal cord is protected by the vertebrae. The brain and spinal cord are both enclosed in the meninges. In central nervous systems, the interneuronal space is filled with a large amount of supporting non-nervous cells called neuroglial cells.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 5.69e-05,
      "usage": {
        "prompt_tokens": 469,
        "completion_tokens": 20,
        "total_tokens": 489,
        "cost": 5.69e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.69e-05,
          "upstream_inference_prompt_cost": 4.69e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930650-O44VJvbOpLwua30Plj2G",
        "object": "chat.completion",
        "created": 1790930650,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 469,
          "completion_tokens": 20,
          "total_tokens": 489,
          "cost": 5.69e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 5.69e-05,
            "upstream_inference_prompt_cost": 4.69e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930650-O44VJvbOpLwua30Plj2G",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.954460790992016,
      "run": "main",
      "record_index": 148
    },
    {
      "case_id": "boolq-dev-1335",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 149,
      "started_at": "2026-10-02T08:44:11.146279+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The central nervous system (CNS) is the part of the nervous system consisting of the brain and spinal cord. The central nervous system is so named because it integrates information it receives from, and coordinates and influences the activity of, all parts of the bodies of bilaterally symmetric animals--that is, all multicellular animals except sponges and radially symmetric animals such as jellyfish--and it contains the majority of the nervous system. Many consider the retina and the optic nerve (cranial nerve II), as well as the olfactory nerves (cranial nerve I) and olfactory epithelium as parts of the CNS, synapsing directly on brain tissue without intermediate ganglia. As such, the olfactory epithelium is the only central nervous tissue in direct contact with the environment, which opens up for therapeutic treatments. The CNS is contained within the dorsal body cavity, with the brain housed in the cranial cavity and the spinal cord in the spinal canal. In vertebrates, the brain is protected by the skull, while the spinal cord is protected by the vertebrae. The brain and spinal cord are both enclosed in the meninges. In central nervous systems, the interneuronal space is filled with a large amount of supporting non-nervous cells called neuroglial cells."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are there nerves in the central nervous system?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.92,
      "valid": true,
      "error": null,
      "cost_usd": 2.5368e-05,
      "usage": {
        "input_tokens": 604,
        "output_tokens": 20,
        "cost": 2.5368e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.92
          }
        },
        "usage": {
          "input_tokens": 604,
          "output_tokens": 20,
          "cost": 2.5368e-05
        },
        "id": "gen-dec-1790930651-G4TeRw77BUa0m5oaZRwp",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930651-G4TeRw77BUa0m5oaZRwp",
      "answer": {
        "type": "noul",
        "noul": 0.92
      },
      "latency_s": 0.38763645800645463,
      "run": "main",
      "record_index": 149
    },
    {
      "case_id": "boolq-dev-1092",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 150,
      "started_at": "2026-10-02T08:44:11.534241+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "New England is a geographical region comprising six states of the northeastern United States: Maine, Vermont, New Hampshire, Massachusetts, Rhode Island, and Connecticut. It is bordered by the state of New York to the west and by the Canadian provinces of New Brunswick and Quebec to the northeast and north, respectively. The Atlantic Ocean is to the east and southeast, and Long Island Sound is to the south. Boston is New England's largest city as well as the capital of Massachusetts. The largest metropolitan area is Greater Boston, which also includes Worcester, Massachusetts (the second-largest city in New England), Manchester, New Hampshire (the largest city in New Hampshire), and Providence, Rhode Island (the capital and largest city of Rhode Island), with nearly a third of the entire region's population."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is new york in the new england region?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.05,
      "valid": true,
      "error": null,
      "cost_usd": 2.0328e-05,
      "usage": {
        "input_tokens": 484,
        "output_tokens": 20,
        "cost": 2.0328e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.05
          }
        },
        "usage": {
          "input_tokens": 484,
          "output_tokens": 20,
          "cost": 2.0328e-05
        },
        "id": "gen-dec-1790930651-XyWzB9Y4JuBlhH2LRVQ4",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930651-XyWzB9Y4JuBlhH2LRVQ4",
      "answer": {
        "type": "noul",
        "noul": 0.05
      },
      "latency_s": 0.22845895899808966,
      "run": "main",
      "record_index": 150
    },
    {
      "case_id": "boolq-dev-1092",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 151,
      "started_at": "2026-10-02T08:44:11.763119+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is new york in the new england region?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"New England is a geographical region comprising six states of the northeastern United States: Maine, Vermont, New Hampshire, Massachusetts, Rhode Island, and Connecticut. It is bordered by the state of New York to the west and by the Canadian provinces of New Brunswick and Quebec to the northeast and north, respectively. The Atlantic Ocean is to the east and southeast, and Long Island Sound is to the south. Boston is New England's largest city as well as the capital of Massachusetts. The largest metropolitan area is Greater Boston, which also includes Worcester, Massachusetts (the second-largest city in New England), Manchester, New Hampshire (the largest city in New Hampshire), and Providence, Rhode Island (the capital and largest city of Rhode Island), with nearly a third of the entire region's population.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.6e-05,
      "usage": {
        "prompt_tokens": 360,
        "completion_tokens": 20,
        "total_tokens": 380,
        "cost": 4.6e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.6e-05,
          "upstream_inference_prompt_cost": 3.6e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930651-1oYr5Wyb0cxZslvMnMdp",
        "object": "chat.completion",
        "created": 1790930651,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 360,
          "completion_tokens": 20,
          "total_tokens": 380,
          "cost": 4.6e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.6e-05,
            "upstream_inference_prompt_cost": 3.6e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930651-1oYr5Wyb0cxZslvMnMdp",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.233808416989632,
      "run": "main",
      "record_index": 151
    },
    {
      "case_id": "policy-023",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 152,
      "started_at": "2026-10-02T08:44:12.997454+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": false,
            "days_since_purchase": 0,
            "defective": false,
            "final_sale": false,
            "price_credits": 247,
            "recalled": true,
            "receipt": false,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.037e-05,
      "usage": {
        "input_tokens": 485,
        "output_tokens": 20,
        "cost": 2.037e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 485,
          "output_tokens": 20,
          "cost": 2.037e-05
        },
        "id": "gen-dec-1790930653-npUhoIN0VvHdOupqR8vK",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930653-npUhoIN0VvHdOupqR8vK",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.25510658402345143,
      "run": "main",
      "record_index": 152
    },
    {
      "case_id": "policy-023",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 153,
      "started_at": "2026-10-02T08:44:13.252992+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": false, \"days_since_purchase\": 0, \"defective\": false, \"final_sale\": false, \"price_credits\": 247, \"recalled\": true, \"receipt\": false, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930653-MK14HIfHwffUvc3OZJiZ",
        "object": "chat.completion",
        "created": 1790930653,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930653-MK14HIfHwffUvc3OZJiZ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.302295250003226,
      "run": "main",
      "record_index": 153
    },
    {
      "case_id": "boolq-dev-2041",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 154,
      "started_at": "2026-10-02T08:44:14.555771+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "An unlisted public company is a public company that is not listed on any stock exchange. Though the criteria vary somewhat between jurisdictions, a public company is a company that is registered as such and generally has a minimum share capital and a minimum number of shareholders. Each stock exchange has its own listing requirements which a company (or other entity) wishing to be listed must meet. Besides not qualifying to be listed, a public company may choose not to be listed on a stock exchange for a number of reasons, including because it is too small to qualify for a stock exchange listing, does not seek public investors, or there are too few shareholders for a listing. There is a cost to the listed entities, in the listing process and ongoing costs as well as in compliance costs such as the maintenance of a company register."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are all public companies listed on the stock exchange?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.03,
      "valid": true,
      "error": null,
      "cost_usd": 2.058e-05,
      "usage": {
        "input_tokens": 490,
        "output_tokens": 20,
        "cost": 2.058e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.03
          }
        },
        "usage": {
          "input_tokens": 490,
          "output_tokens": 20,
          "cost": 2.058e-05
        },
        "id": "gen-dec-1790930654-bzCHseniZ2DIly3oTbgo",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930654-bzCHseniZ2DIly3oTbgo",
      "answer": {
        "type": "noul",
        "noul": 0.03
      },
      "latency_s": 0.2360435409937054,
      "run": "main",
      "record_index": 154
    },
    {
      "case_id": "boolq-dev-2041",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 155,
      "started_at": "2026-10-02T08:44:14.792185+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are all public companies listed on the stock exchange?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"An unlisted public company is a public company that is not listed on any stock exchange. Though the criteria vary somewhat between jurisdictions, a public company is a company that is registered as such and generally has a minimum share capital and a minimum number of shareholders. Each stock exchange has its own listing requirements which a company (or other entity) wishing to be listed must meet. Besides not qualifying to be listed, a public company may choose not to be listed on a stock exchange for a number of reasons, including because it is too small to qualify for a stock exchange listing, does not seek public investors, or there are too few shareholders for a listing. There is a cost to the listed entities, in the listing process and ongoing costs as well as in compliance costs such as the maintenance of a company register.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.67e-05,
      "usage": {
        "prompt_tokens": 367,
        "completion_tokens": 20,
        "total_tokens": 387,
        "cost": 4.67e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.67e-05,
          "upstream_inference_prompt_cost": 3.67e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930654-h34ALIXaPDZ8PODKFG7q",
        "object": "chat.completion",
        "created": 1790930654,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 367,
          "completion_tokens": 20,
          "total_tokens": 387,
          "cost": 4.67e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.67e-05,
            "upstream_inference_prompt_cost": 3.67e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930654-h34ALIXaPDZ8PODKFG7q",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.064374375011539,
      "run": "main",
      "record_index": 155
    },
    {
      "case_id": "probability-038",
      "experiment": "probability",
      "target": 0.0,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 156,
      "started_at": "2026-10-02T08:44:15.857054+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Are both selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 1, \"unmarked\": 5}, \"event\": \"both_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.74e-05,
      "usage": {
        "prompt_tokens": 274,
        "completion_tokens": 20,
        "total_tokens": 294,
        "cost": 3.74e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.74e-05,
          "upstream_inference_prompt_cost": 2.74e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930655-4KFaapq5YIFe99CgQ6NL",
        "object": "chat.completion",
        "created": 1790930655,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 274,
          "completion_tokens": 20,
          "total_tokens": 294,
          "cost": 3.74e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.74e-05,
            "upstream_inference_prompt_cost": 2.74e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930655-4KFaapq5YIFe99CgQ6NL",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9810665829863865,
      "run": "main",
      "record_index": 156
    },
    {
      "case_id": "probability-038",
      "experiment": "probability",
      "target": 0.0,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 157,
      "started_at": "2026-10-02T08:44:16.838587+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 1,
            "unmarked": 5
          },
          "event": "both_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Are both selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.05,
      "valid": true,
      "error": null,
      "cost_usd": 1.7262e-05,
      "usage": {
        "input_tokens": 411,
        "output_tokens": 20,
        "cost": 1.7262e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.05
          }
        },
        "usage": {
          "input_tokens": 411,
          "output_tokens": 20,
          "cost": 1.7262e-05
        },
        "id": "gen-dec-1790930656-MZQAoC0FKv8sswkFEqjL",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930656-MZQAoC0FKv8sswkFEqjL",
      "answer": {
        "type": "noul",
        "noul": 0.05
      },
      "latency_s": 0.2268226670275908,
      "run": "main",
      "record_index": 157
    },
    {
      "case_id": "policy-057",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 158,
      "started_at": "2026-10-02T08:44:17.065857+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 1501, \"active_promotion\": false, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": false, \"member\": false, \"subtotal_credits\": 101, \"weight_kg\": 20}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930657-T9wo6JXt04gDkpHQZ337",
        "object": "chat.completion",
        "created": 1790930657,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930657-T9wo6JXt04gDkpHQZ337",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2716436669870745,
      "run": "main",
      "record_index": 158
    },
    {
      "case_id": "policy-057",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 159,
      "started_at": "2026-10-02T08:44:18.338848+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 1501,
            "active_promotion": false,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": false,
            "member": false,
            "subtotal_credits": 101,
            "weight_kg": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.07,
      "valid": true,
      "error": null,
      "cost_usd": 2.1336e-05,
      "usage": {
        "input_tokens": 508,
        "output_tokens": 20,
        "cost": 2.1336e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.07
          }
        },
        "usage": {
          "input_tokens": 508,
          "output_tokens": 20,
          "cost": 2.1336e-05
        },
        "id": "gen-dec-1790930658-tmE5n0m5a3hCMTPl1jSy",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930658-tmE5n0m5a3hCMTPl1jSy",
      "answer": {
        "type": "noul",
        "noul": 0.07
      },
      "latency_s": 0.22602383300545625,
      "run": "main",
      "record_index": 159
    },
    {
      "case_id": "probability-016",
      "experiment": "probability",
      "target": 0.023715415019762844,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 160,
      "started_at": "2026-10-02T08:44:18.565668+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Are both selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 4, \"unmarked\": 19}, \"event\": \"both_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.023715415019762844,
      "valid": true,
      "error": null,
      "cost_usd": 4.09e-05,
      "usage": {
        "prompt_tokens": 274,
        "completion_tokens": 27,
        "total_tokens": 301,
        "cost": 4.09e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.09e-05,
          "upstream_inference_prompt_cost": 2.74e-05,
          "upstream_inference_completions_cost": 1.35e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930658-tu1qYDZzTCBGFjmiC0GA",
        "object": "chat.completion",
        "created": 1790930658,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.023715415019762844}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 274,
          "completion_tokens": 27,
          "total_tokens": 301,
          "cost": 4.09e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.09e-05,
            "upstream_inference_prompt_cost": 2.74e-05,
            "upstream_inference_completions_cost": 1.35e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930658-tu1qYDZzTCBGFjmiC0GA",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.023715415019762844
      },
      "latency_s": 1.0959417920093983,
      "run": "main",
      "record_index": 160
    },
    {
      "case_id": "probability-016",
      "experiment": "probability",
      "target": 0.023715415019762844,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 161,
      "started_at": "2026-10-02T08:44:19.662040+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 4,
            "unmarked": 19
          },
          "event": "both_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Are both selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.15,
      "valid": true,
      "error": null,
      "cost_usd": 1.7304e-05,
      "usage": {
        "input_tokens": 412,
        "output_tokens": 20,
        "cost": 1.7304e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.15
          }
        },
        "usage": {
          "input_tokens": 412,
          "output_tokens": 20,
          "cost": 1.7304e-05
        },
        "id": "gen-dec-1790930659-XI8paktHjP4R8SIHmTpO",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930659-XI8paktHjP4R8SIHmTpO",
      "answer": {
        "type": "noul",
        "noul": 0.15
      },
      "latency_s": 0.23366291701677255,
      "run": "main",
      "record_index": 161
    },
    {
      "case_id": "boolq-dev-0525",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 162,
      "started_at": "2026-10-02T08:44:19.896389+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The Canadian dollar (symbol: $; code: CAD; French: dollar canadien) is the currency of Canada. It is abbreviated with the dollar sign $, or sometimes Can$ or C$ to distinguish it from other dollar-denominated currencies. It is divided into 100 cents (¢)."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are canadian dollars the same as american dollars?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.06,
      "valid": true,
      "error": null,
      "cost_usd": 1.6422e-05,
      "usage": {
        "input_tokens": 391,
        "output_tokens": 20,
        "cost": 1.6422e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.06
          }
        },
        "usage": {
          "input_tokens": 391,
          "output_tokens": 20,
          "cost": 1.6422e-05
        },
        "id": "gen-dec-1790930659-OLlV6WwS6odY5FMglCGe",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930659-OLlV6WwS6odY5FMglCGe",
      "answer": {
        "type": "noul",
        "noul": 0.06
      },
      "latency_s": 0.23506987502332777,
      "run": "main",
      "record_index": 162
    },
    {
      "case_id": "boolq-dev-0525",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 163,
      "started_at": "2026-10-02T08:44:20.131873+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are canadian dollars the same as american dollars?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The Canadian dollar (symbol: $; code: CAD; French: dollar canadien) is the currency of Canada. It is abbreviated with the dollar sign $, or sometimes Can$ or C$ to distinguish it from other dollar-denominated currencies. It is divided into 100 cents (¢).\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.64e-05,
      "usage": {
        "prompt_tokens": 264,
        "completion_tokens": 20,
        "total_tokens": 284,
        "cost": 3.64e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.64e-05,
          "upstream_inference_prompt_cost": 2.64e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930660-yizyltjfwComlkmXgUnp",
        "object": "chat.completion",
        "created": 1790930660,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 264,
          "completion_tokens": 20,
          "total_tokens": 284,
          "cost": 3.64e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.64e-05,
            "upstream_inference_prompt_cost": 2.64e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930660-yizyltjfwComlkmXgUnp",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0164400830108207,
      "run": "main",
      "record_index": 163
    },
    {
      "case_id": "boolq-dev-0980",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 164,
      "started_at": "2026-10-02T08:44:21.149710+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is pyruvic acid and pyruvate the same thing?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Pyruvic acid (CHCOCOOH) is the simplest of the alpha-keto acids, with a carboxylic acid and a ketone functional group. Pyruvate (/paɪˈruːveɪt/), the conjugate base, CHCOCOO, is a key intermediate in several metabolic pathways.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.8e-05,
      "usage": {
        "prompt_tokens": 280,
        "completion_tokens": 20,
        "total_tokens": 300,
        "cost": 3.8e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.8e-05,
          "upstream_inference_prompt_cost": 2.8e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930661-JspKyMXFv8duHn6BMzGl",
        "object": "chat.completion",
        "created": 1790930661,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 280,
          "completion_tokens": 20,
          "total_tokens": 300,
          "cost": 3.8e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.8e-05,
            "upstream_inference_prompt_cost": 2.8e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930661-JspKyMXFv8duHn6BMzGl",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.1693377909832634,
      "run": "main",
      "record_index": 164
    },
    {
      "case_id": "boolq-dev-0980",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 165,
      "started_at": "2026-10-02T08:44:22.319623+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Pyruvic acid (CHCOCOOH) is the simplest of the alpha-keto acids, with a carboxylic acid and a ketone functional group. Pyruvate (/paɪˈruːveɪt/), the conjugate base, CHCOCOO, is a key intermediate in several metabolic pathways."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is pyruvic acid and pyruvate the same thing?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.11,
      "valid": true,
      "error": null,
      "cost_usd": 1.6968e-05,
      "usage": {
        "input_tokens": 404,
        "output_tokens": 20,
        "cost": 1.6968e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.11
          }
        },
        "usage": {
          "input_tokens": 404,
          "output_tokens": 20,
          "cost": 1.6968e-05
        },
        "id": "gen-dec-1790930662-wlzVKxElWlrusiY1m9wa",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930662-wlzVKxElWlrusiY1m9wa",
      "answer": {
        "type": "noul",
        "noul": 0.11
      },
      "latency_s": 0.21495170900016092,
      "run": "main",
      "record_index": 165
    },
    {
      "case_id": "policy-060",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 166,
      "started_at": "2026-10-02T08:44:22.535705+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 1863,
            "active_promotion": true,
            "gift_wrapping": false,
            "hazardous": false,
            "mainland": true,
            "member": true,
            "subtotal_credits": 100,
            "weight_kg": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1336e-05,
      "usage": {
        "input_tokens": 508,
        "output_tokens": 20,
        "cost": 2.1336e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 508,
          "output_tokens": 20,
          "cost": 2.1336e-05
        },
        "id": "gen-dec-1790930662-X0C5IGKTKrAmLHEPOqjs",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930662-X0C5IGKTKrAmLHEPOqjs",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.23865200000000186,
      "run": "main",
      "record_index": 166
    },
    {
      "case_id": "policy-060",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 167,
      "started_at": "2026-10-02T08:44:22.775143+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 1863, \"active_promotion\": true, \"gift_wrapping\": false, \"hazardous\": false, \"mainland\": true, \"member\": true, \"subtotal_credits\": 100, \"weight_kg\": 20}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930662-aSElbsOPB41ZoDCjZBhQ",
        "object": "chat.completion",
        "created": 1790930662,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930662-aSElbsOPB41ZoDCjZBhQ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9869363330071792,
      "run": "main",
      "record_index": 167
    },
    {
      "case_id": "policy-051",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 168,
      "started_at": "2026-10-02T08:44:23.762582+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 90, \"defective\": true, \"final_sale\": true, \"price_credits\": 171, \"recalled\": false, \"receipt\": true, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930663-oB5n8br6ovKzu34RCN3i",
        "object": "chat.completion",
        "created": 1790930663,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930663-oB5n8br6ovKzu34RCN3i",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2400692499941215,
      "run": "main",
      "record_index": 168
    },
    {
      "case_id": "policy-051",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 169,
      "started_at": "2026-10-02T08:44:25.003720+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 90,
            "defective": true,
            "final_sale": true,
            "price_credits": 171,
            "recalled": false,
            "receipt": true,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.11,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.11
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930665-xPb4CjN0tH7CaEp6V9uL",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930665-xPb4CjN0tH7CaEp6V9uL",
      "answer": {
        "type": "noul",
        "noul": 0.11
      },
      "latency_s": 0.31808541700593196,
      "run": "main",
      "record_index": 169
    },
    {
      "case_id": "probability-060",
      "experiment": "probability",
      "target": 0.4957208689927584,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 170,
      "started_at": "2026-10-02T08:44:25.322235+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 6, \"B\": 3}, \"boxes\": {\"A\": {\"marked\": 23, \"unmarked\": 26}, \"B\": {\"marked\": 17, \"unmarked\": 14}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.4467,
      "valid": true,
      "error": null,
      "cost_usd": 4.44e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 23,
        "total_tokens": 352,
        "cost": 4.44e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.44e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930665-tu3lm7B7dFYVmwpfJHrM",
        "object": "chat.completion",
        "created": 1790930665,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.4467}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 23,
          "total_tokens": 352,
          "cost": 4.44e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.44e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930665-tu3lm7B7dFYVmwpfJHrM",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.4467
      },
      "latency_s": 1.5809830829966813,
      "run": "main",
      "record_index": 170
    },
    {
      "case_id": "probability-060",
      "experiment": "probability",
      "target": 0.4957208689927584,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 171,
      "started_at": "2026-10-02T08:44:26.903690+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 6,
            "B": 3
          },
          "boxes": {
            "A": {
              "marked": 23,
              "unmarked": 26
            },
            "B": {
              "marked": 17,
              "unmarked": 14
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.53,
      "valid": true,
      "error": null,
      "cost_usd": 2.0076e-05,
      "usage": {
        "input_tokens": 478,
        "output_tokens": 20,
        "cost": 2.0076e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.53
          }
        },
        "usage": {
          "input_tokens": 478,
          "output_tokens": 20,
          "cost": 2.0076e-05
        },
        "id": "gen-dec-1790930666-N9IKyoRLvq3dWlECYWuu",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930666-N9IKyoRLvq3dWlECYWuu",
      "answer": {
        "type": "noul",
        "noul": 0.53
      },
      "latency_s": 0.21743774999049492,
      "run": "main",
      "record_index": 171
    },
    {
      "case_id": "probability-015",
      "experiment": "probability",
      "target": 0.24516129032258063,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 172,
      "started_at": "2026-10-02T08:44:27.121529+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is at least one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 4, \"unmarked\": 27}, \"event\": \"at_least_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.2426,
      "valid": true,
      "error": null,
      "cost_usd": 3.97e-05,
      "usage": {
        "prompt_tokens": 282,
        "completion_tokens": 23,
        "total_tokens": 305,
        "cost": 3.97e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.97e-05,
          "upstream_inference_prompt_cost": 2.82e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930667-mWcMPRvhRJE4PdZq1isS",
        "object": "chat.completion",
        "created": 1790930667,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.2426}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 282,
          "completion_tokens": 23,
          "total_tokens": 305,
          "cost": 3.97e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.97e-05,
            "upstream_inference_prompt_cost": 2.82e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930667-mWcMPRvhRJE4PdZq1isS",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.2426
      },
      "latency_s": 1.121371667017229,
      "run": "main",
      "record_index": 172
    },
    {
      "case_id": "probability-015",
      "experiment": "probability",
      "target": 0.24516129032258063,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 173,
      "started_at": "2026-10-02T08:44:28.243526+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 4,
            "unmarked": 27
          },
          "event": "at_least_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is at least one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.59,
      "valid": true,
      "error": null,
      "cost_usd": 1.764e-05,
      "usage": {
        "input_tokens": 420,
        "output_tokens": 20,
        "cost": 1.764e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.59
          }
        },
        "usage": {
          "input_tokens": 420,
          "output_tokens": 20,
          "cost": 1.764e-05
        },
        "id": "gen-dec-1790930668-jt0CqYMVuTtHcIWfScNt",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930668-jt0CqYMVuTtHcIWfScNt",
      "answer": {
        "type": "noul",
        "noul": 0.59
      },
      "latency_s": 0.28803825000068173,
      "run": "main",
      "record_index": 173
    },
    {
      "case_id": "policy-052",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 174,
      "started_at": "2026-10-02T08:44:28.531962+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "green",
            "clearance": 3,
            "emergency_permit": true,
            "experience_years": 27,
            "role": "staff",
            "sponsor": true,
            "suspended": false,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930668-Su2VR13DiDidFVEkFPbg",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930668-Su2VR13DiDidFVEkFPbg",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.26521466599660926,
      "run": "main",
      "record_index": 174
    },
    {
      "case_id": "policy-052",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 175,
      "started_at": "2026-10-02T08:44:28.797720+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"green\", \"clearance\": 3, \"emergency_permit\": true, \"experience_years\": 27, \"role\": \"staff\", \"sponsor\": true, \"suspended\": false, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930668-1nDjO3E9U2wKOVwpmgm6",
        "object": "chat.completion",
        "created": 1790930668,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930668-1nDjO3E9U2wKOVwpmgm6",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9410200419952162,
      "run": "main",
      "record_index": 175
    },
    {
      "case_id": "policy-019",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 176,
      "started_at": "2026-10-02T08:44:29.739316+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 856, \"active_promotion\": false, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": true, \"member\": true, \"subtotal_credits\": 100, \"weight_kg\": 19}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.64e-05,
      "usage": {
        "prompt_tokens": 364,
        "completion_tokens": 20,
        "total_tokens": 384,
        "cost": 4.64e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.64e-05,
          "upstream_inference_prompt_cost": 3.64e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930669-Xqslr0gBLKTqK9KSMJPP",
        "object": "chat.completion",
        "created": 1790930669,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 364,
          "completion_tokens": 20,
          "total_tokens": 384,
          "cost": 4.64e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.64e-05,
            "upstream_inference_prompt_cost": 3.64e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930669-Xqslr0gBLKTqK9KSMJPP",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0078743750054855,
      "run": "main",
      "record_index": 176
    },
    {
      "case_id": "policy-019",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 177,
      "started_at": "2026-10-02T08:44:30.748586+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 856,
            "active_promotion": false,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": true,
            "member": true,
            "subtotal_credits": 100,
            "weight_kg": 19
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930670-XOFDv9BiMKRl6lyj5Q0V",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930670-XOFDv9BiMKRl6lyj5Q0V",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.24609120801324025,
      "run": "main",
      "record_index": 177
    },
    {
      "case_id": "boolq-dev-2217",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 178,
      "started_at": "2026-10-02T08:44:30.995592+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "A Bank State Branch (often referred to as ``BSB'') is the name used in Australia for a bank code, which is a branch identifier. The BSB is normally used in association with the account number system used by each financial institution. The structure of the BSB + account number does not permit for account numbers to be transferable between financial institutions. While similar in structure, the New Zealand and Australian systems are only used in domestic transactions and are incompatible with each other. For international transfers, a SWIFT code is used in addition to the BSB and account number."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are swift codes and bsb numbers the same?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.04,
      "valid": true,
      "error": null,
      "cost_usd": 1.8774e-05,
      "usage": {
        "input_tokens": 447,
        "output_tokens": 20,
        "cost": 1.8774e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.04
          }
        },
        "usage": {
          "input_tokens": 447,
          "output_tokens": 20,
          "cost": 1.8774e-05
        },
        "id": "gen-dec-1790930671-j3XR20cHcGUzzqKWGJ3G",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930671-j3XR20cHcGUzzqKWGJ3G",
      "answer": {
        "type": "noul",
        "noul": 0.04
      },
      "latency_s": 0.22258908400544897,
      "run": "main",
      "record_index": 178
    },
    {
      "case_id": "boolq-dev-2217",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 179,
      "started_at": "2026-10-02T08:44:31.219451+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are swift codes and bsb numbers the same?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"A Bank State Branch (often referred to as ``BSB'') is the name used in Australia for a bank code, which is a branch identifier. The BSB is normally used in association with the account number system used by each financial institution. The structure of the BSB + account number does not permit for account numbers to be transferable between financial institutions. While similar in structure, the New Zealand and Australian systems are only used in domestic transactions and are incompatible with each other. For international transfers, a SWIFT code is used in addition to the BSB and account number.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.22e-05,
      "usage": {
        "prompt_tokens": 322,
        "completion_tokens": 20,
        "total_tokens": 342,
        "cost": 4.22e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.22e-05,
          "upstream_inference_prompt_cost": 3.22e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930671-VBjTKplvYRPM1qYFA4yf",
        "object": "chat.completion",
        "created": 1790930671,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 322,
          "completion_tokens": 20,
          "total_tokens": 342,
          "cost": 4.22e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.22e-05,
            "upstream_inference_prompt_cost": 3.22e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930671-VBjTKplvYRPM1qYFA4yf",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9502893330063671,
      "run": "main",
      "record_index": 179
    },
    {
      "case_id": "boolq-dev-2011",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 180,
      "started_at": "2026-10-02T08:44:32.170272+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"did dwight howard play for the houston rockets?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Dwight David Howard (born December 8, 1985) is an American professional basketball player for the Washington Wizards of the National Basketball Association (NBA). Howard, who plays center, spent his high school career at Southwest Atlanta Christian Academy. He chose to forgo college and entered the 2004 NBA draft, and was selected first overall by the Orlando Magic. An eight-time All-Star, eight-time All-NBA Team honoree, five-time All-Defensive member, and three-time Defensive Player of the Year, Howard has been ranked consistently as one of the best in the league in rebounds, blocks, and field goal percentage, and has set numerous franchise and league records. In 2009, he led the Magic to the NBA Finals. In 2012, after eight seasons with the Magic, Howard was traded to the Los Angeles Lakers. After one season with the Lakers, he joined the Houston Rockets. He spent three seasons with the Rockets, before joining the Atlanta Hawks in 2016, and being traded to the Hornets in 2017. In July 2018, he was traded once again, this time to the Brooklyn Nets.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 5.43e-05,
      "usage": {
        "prompt_tokens": 443,
        "completion_tokens": 20,
        "total_tokens": 463,
        "cost": 5.43e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.43e-05,
          "upstream_inference_prompt_cost": 4.43e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930672-qGIY9NbZ6WEP3fc2iLev",
        "object": "chat.completion",
        "created": 1790930672,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 443,
          "completion_tokens": 20,
          "total_tokens": 463,
          "cost": 5.43e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 5.43e-05,
            "upstream_inference_prompt_cost": 4.43e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930672-qGIY9NbZ6WEP3fc2iLev",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9718628330156207,
      "run": "main",
      "record_index": 180
    },
    {
      "case_id": "boolq-dev-2011",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 181,
      "started_at": "2026-10-02T08:44:33.142595+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Dwight David Howard (born December 8, 1985) is an American professional basketball player for the Washington Wizards of the National Basketball Association (NBA). Howard, who plays center, spent his high school career at Southwest Atlanta Christian Academy. He chose to forgo college and entered the 2004 NBA draft, and was selected first overall by the Orlando Magic. An eight-time All-Star, eight-time All-NBA Team honoree, five-time All-Defensive member, and three-time Defensive Player of the Year, Howard has been ranked consistently as one of the best in the league in rebounds, blocks, and field goal percentage, and has set numerous franchise and league records. In 2009, he led the Magic to the NBA Finals. In 2012, after eight seasons with the Magic, Howard was traded to the Los Angeles Lakers. After one season with the Lakers, he joined the Houston Rockets. He spent three seasons with the Rockets, before joining the Atlanta Hawks in 2016, and being traded to the Hornets in 2017. In July 2018, he was traded once again, this time to the Brooklyn Nets."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "did dwight howard play for the houston rockets?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 2.4402e-05,
      "usage": {
        "input_tokens": 581,
        "output_tokens": 20,
        "cost": 2.4402e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 581,
          "output_tokens": 20,
          "cost": 2.4402e-05
        },
        "id": "gen-dec-1790930673-KJoD7lwxwMwZ8BhCggdB",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930673-KJoD7lwxwMwZ8BhCggdB",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.22267533300328068,
      "run": "main",
      "record_index": 181
    },
    {
      "case_id": "probability-010",
      "experiment": "probability",
      "target": 0.5,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 182,
      "started_at": "2026-10-02T08:44:33.365706+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 8,
              "unmarked": 29
            },
            "B": {
              "marked": 20,
              "unmarked": 20
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.51,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.51
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930673-twcgAnssuX29XByH1Yo5",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930673-twcgAnssuX29XByH1Yo5",
      "answer": {
        "type": "noul",
        "noul": 0.51
      },
      "latency_s": 0.22714366600848734,
      "run": "main",
      "record_index": 182
    },
    {
      "case_id": "probability-010",
      "experiment": "probability",
      "target": 0.5,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 183,
      "started_at": "2026-10-02T08:44:33.593273+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 8, \"unmarked\": 29}, \"B\": {\"marked\": 20, \"unmarked\": 20}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930673-of9k8fXrKmKDPGxlUjpy",
        "object": "chat.completion",
        "created": 1790930673,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930673-of9k8fXrKmKDPGxlUjpy",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5
      },
      "latency_s": 1.4065097909769975,
      "run": "main",
      "record_index": 183
    },
    {
      "case_id": "boolq-dev-0877",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 184,
      "started_at": "2026-10-02T08:44:35.000363+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "A federal government is the common or national government of a federation. The United States is considered the first modern federation. After declaring independence from Britain, the U.S. adopted its first constitution, the Articles of Confederation in 1781. This was the first step towards federalism by establishing the confederal Congress. However, Congress was limited as to its ability to pursue economic, military, and judiciary reform. In 1787, a Constitutional Convention drafted the United States Constitution during the Philadelphia Convention. After the ratification of the Constitution by nine states in 1788, the U.S. was officially a federation, putting the U.S. in a unique position where the central government exists by the sufferance of the individual states rather than the reverse."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is the national government the same as the federal government?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.93,
      "valid": true,
      "error": null,
      "cost_usd": 2.0706e-05,
      "usage": {
        "input_tokens": 493,
        "output_tokens": 20,
        "cost": 2.0706e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.93
          }
        },
        "usage": {
          "input_tokens": 493,
          "output_tokens": 20,
          "cost": 2.0706e-05
        },
        "id": "gen-dec-1790930675-uEQUmKhlUjLRD0Mz7KQ5",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930675-uEQUmKhlUjLRD0Mz7KQ5",
      "answer": {
        "type": "noul",
        "noul": 0.93
      },
      "latency_s": 0.27408166602253914,
      "run": "main",
      "record_index": 184
    },
    {
      "case_id": "boolq-dev-0877",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 185,
      "started_at": "2026-10-02T08:44:35.274902+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is the national government the same as the federal government?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"A federal government is the common or national government of a federation. The United States is considered the first modern federation. After declaring independence from Britain, the U.S. adopted its first constitution, the Articles of Confederation in 1781. This was the first step towards federalism by establishing the confederal Congress. However, Congress was limited as to its ability to pursue economic, military, and judiciary reform. In 1787, a Constitutional Convention drafted the United States Constitution during the Philadelphia Convention. After the ratification of the Constitution by nine states in 1788, the U.S. was officially a federation, putting the U.S. in a unique position where the central government exists by the sufferance of the individual states rather than the reverse.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.58e-05,
      "usage": {
        "prompt_tokens": 358,
        "completion_tokens": 20,
        "total_tokens": 378,
        "cost": 4.58e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.58e-05,
          "upstream_inference_prompt_cost": 3.58e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930675-PmrgP7P42xtOZ2K1jgqH",
        "object": "chat.completion",
        "created": 1790930675,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 358,
          "completion_tokens": 20,
          "total_tokens": 378,
          "cost": 4.58e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.58e-05,
            "upstream_inference_prompt_cost": 3.58e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930675-PmrgP7P42xtOZ2K1jgqH",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.017610084003536,
      "run": "main",
      "record_index": 185
    },
    {
      "case_id": "policy-032",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 186,
      "started_at": "2026-10-02T08:44:36.292987+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 1676,
            "active_promotion": false,
            "gift_wrapping": false,
            "hazardous": false,
            "mainland": true,
            "member": false,
            "subtotal_credits": 100,
            "weight_kg": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1336e-05,
      "usage": {
        "input_tokens": 508,
        "output_tokens": 20,
        "cost": 2.1336e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 508,
          "output_tokens": 20,
          "cost": 2.1336e-05
        },
        "id": "gen-dec-1790930676-ikNVDrN7SYfq3sHUNH2R",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930676-ikNVDrN7SYfq3sHUNH2R",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.248715583991725,
      "run": "main",
      "record_index": 186
    },
    {
      "case_id": "policy-032",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 187,
      "started_at": "2026-10-02T08:44:36.542182+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 1676, \"active_promotion\": false, \"gift_wrapping\": false, \"hazardous\": false, \"mainland\": true, \"member\": false, \"subtotal_credits\": 100, \"weight_kg\": 20}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930676-ejfZ9yGm1ZuqXJFIj4Py",
        "object": "chat.completion",
        "created": 1790930676,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930676-ejfZ9yGm1ZuqXJFIj4Py",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1132097080117092,
      "run": "main",
      "record_index": 187
    },
    {
      "case_id": "policy-017",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 188,
      "started_at": "2026-10-02T08:44:37.655896+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": false,
            "days_since_purchase": 0,
            "defective": true,
            "final_sale": true,
            "price_credits": 201,
            "recalled": true,
            "receipt": true,
            "unused": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.037e-05,
      "usage": {
        "input_tokens": 485,
        "output_tokens": 20,
        "cost": 2.037e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 485,
          "output_tokens": 20,
          "cost": 2.037e-05
        },
        "id": "gen-dec-1790930677-tJXGki331ruDgrkk6NUm",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930677-tJXGki331ruDgrkk6NUm",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.2680019160034135,
      "run": "main",
      "record_index": 188
    },
    {
      "case_id": "policy-017",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 189,
      "started_at": "2026-10-02T08:44:37.924315+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": false, \"days_since_purchase\": 0, \"defective\": true, \"final_sale\": true, \"price_credits\": 201, \"recalled\": true, \"receipt\": true, \"unused\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930677-6BxSXikNjTXVK3XOz1LU",
        "object": "chat.completion",
        "created": 1790930677,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930677-6BxSXikNjTXVK3XOz1LU",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 2.3139472500188276,
      "run": "main",
      "record_index": 189
    },
    {
      "case_id": "policy-008",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 190,
      "started_at": "2026-10-02T08:44:40.238749+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 450,
            "active_promotion": false,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": true,
            "member": false,
            "subtotal_credits": 100,
            "weight_kg": 21
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.06,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.06
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930680-7bB700TSGfTkOClStLQ0",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930680-7bB700TSGfTkOClStLQ0",
      "answer": {
        "type": "noul",
        "noul": 0.06
      },
      "latency_s": 0.2613263750099577,
      "run": "main",
      "record_index": 190
    },
    {
      "case_id": "policy-008",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 191,
      "started_at": "2026-10-02T08:44:40.500462+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 450, \"active_promotion\": false, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": true, \"member\": false, \"subtotal_credits\": 100, \"weight_kg\": 21}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.64e-05,
      "usage": {
        "prompt_tokens": 364,
        "completion_tokens": 20,
        "total_tokens": 384,
        "cost": 4.64e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.64e-05,
          "upstream_inference_prompt_cost": 3.64e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930680-7I4HtT1DIFSac18Fc8VX",
        "object": "chat.completion",
        "created": 1790930680,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 364,
          "completion_tokens": 20,
          "total_tokens": 384,
          "cost": 4.64e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.64e-05,
            "upstream_inference_prompt_cost": 3.64e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930680-7I4HtT1DIFSac18Fc8VX",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0064754580089357,
      "run": "main",
      "record_index": 191
    },
    {
      "case_id": "boolq-dev-0065",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 192,
      "started_at": "2026-10-02T08:44:41.507397+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "An acute triangle is a triangle with all three angles acute (less than 90°). An obtuse triangle is one with one obtuse angle (greater than 90°) and two acute angles. Since a triangle's angles must sum to 180°, no triangle can have more than one obtuse angle."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "do all triangles have at least two acute angles?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.96,
      "valid": true,
      "error": null,
      "cost_usd": 1.6632e-05,
      "usage": {
        "input_tokens": 396,
        "output_tokens": 20,
        "cost": 1.6632e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.96
          }
        },
        "usage": {
          "input_tokens": 396,
          "output_tokens": 20,
          "cost": 1.6632e-05
        },
        "id": "gen-dec-1790930681-sFGBwBZYF5kVUXQ6Oy1k",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930681-sFGBwBZYF5kVUXQ6Oy1k",
      "answer": {
        "type": "noul",
        "noul": 0.96
      },
      "latency_s": 0.38075420798850246,
      "run": "main",
      "record_index": 192
    },
    {
      "case_id": "boolq-dev-0065",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 193,
      "started_at": "2026-10-02T08:44:41.888615+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"do all triangles have at least two acute angles?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"An acute triangle is a triangle with all three angles acute (less than 90°). An obtuse triangle is one with one obtuse angle (greater than 90°) and two acute angles. Since a triangle's angles must sum to 180°, no triangle can have more than one obtuse angle.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.69e-05,
      "usage": {
        "prompt_tokens": 269,
        "completion_tokens": 20,
        "total_tokens": 289,
        "cost": 3.69e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.69e-05,
          "upstream_inference_prompt_cost": 2.69e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930681-T36KZA1Z3PMVvtL2YjH8",
        "object": "chat.completion",
        "created": 1790930681,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 269,
          "completion_tokens": 20,
          "total_tokens": 289,
          "cost": 3.69e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.69e-05,
            "upstream_inference_prompt_cost": 2.69e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930681-T36KZA1Z3PMVvtL2YjH8",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1039640000090003,
      "run": "main",
      "record_index": 193
    },
    {
      "case_id": "probability-045",
      "experiment": "probability",
      "target": 0.5,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 194,
      "started_at": "2026-10-02T08:44:42.993378+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 20, \"unmarked\": 20}, \"B\": {\"marked\": 2, \"unmarked\": 54}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"A\", \"selection\": \"Choose one item uniformly at random from group A only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930683-bTg5cGxWzkVyVdpf3CFh",
        "object": "chat.completion",
        "created": 1790930683,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930683-bTg5cGxWzkVyVdpf3CFh",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5
      },
      "latency_s": 0.9976881670008879,
      "run": "main",
      "record_index": 194
    },
    {
      "case_id": "probability-045",
      "experiment": "probability",
      "target": 0.5,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 195,
      "started_at": "2026-10-02T08:44:43.991572+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 20,
              "unmarked": 20
            },
            "B": {
              "marked": 2,
              "unmarked": 54
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "A",
          "selection": "Choose one item uniformly at random from group A only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.51,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.51
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930684-wu5aZPdWl0yIZzoWf8BK",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930684-wu5aZPdWl0yIZzoWf8BK",
      "answer": {
        "type": "noul",
        "noul": 0.51
      },
      "latency_s": 0.25186120899161324,
      "run": "main",
      "record_index": 195
    },
    {
      "case_id": "probability-046",
      "experiment": "probability",
      "target": 0.14129919393077287,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 196,
      "started_at": "2026-10-02T08:44:44.243814+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 2, \"B\": 4}, \"boxes\": {\"A\": {\"marked\": 4, \"unmarked\": 33}, \"B\": {\"marked\": 3, \"unmarked\": 16}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0851,
      "valid": true,
      "error": null,
      "cost_usd": 4.44e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 23,
        "total_tokens": 352,
        "cost": 4.44e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.44e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930684-zn4XNyWjlG6BqqzLtyVI",
        "object": "chat.completion",
        "created": 1790930684,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.0851}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 23,
          "total_tokens": 352,
          "cost": 4.44e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.44e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930684-zn4XNyWjlG6BqqzLtyVI",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.0851
      },
      "latency_s": 1.257500459003495,
      "run": "main",
      "record_index": 196
    },
    {
      "case_id": "probability-046",
      "experiment": "probability",
      "target": 0.14129919393077287,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 197,
      "started_at": "2026-10-02T08:44:45.502494+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 2,
            "B": 4
          },
          "boxes": {
            "A": {
              "marked": 4,
              "unmarked": 33
            },
            "B": {
              "marked": 3,
              "unmarked": 16
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.15,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.15
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930685-8KSQ6UTMGgVNU8zgac5V",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930685-8KSQ6UTMGgVNU8zgac5V",
      "answer": {
        "type": "noul",
        "noul": 0.15
      },
      "latency_s": 0.24772379201021977,
      "run": "main",
      "record_index": 197
    },
    {
      "case_id": "probability-001",
      "experiment": "probability",
      "target": 0.5,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 198,
      "started_at": "2026-10-02T08:44:45.750651+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 11,
              "unmarked": 30
            },
            "B": {
              "marked": 15,
              "unmarked": 15
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.53,
      "valid": true,
      "error": null,
      "cost_usd": 1.848e-05,
      "usage": {
        "input_tokens": 440,
        "output_tokens": 20,
        "cost": 1.848e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.53
          }
        },
        "usage": {
          "input_tokens": 440,
          "output_tokens": 20,
          "cost": 1.848e-05
        },
        "id": "gen-dec-1790930685-rtZGoyPdft2NU9wTda4W",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930685-rtZGoyPdft2NU9wTda4W",
      "answer": {
        "type": "noul",
        "noul": 0.53
      },
      "latency_s": 0.28320012500626035,
      "run": "main",
      "record_index": 198
    },
    {
      "case_id": "probability-001",
      "experiment": "probability",
      "target": 0.5,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 199,
      "started_at": "2026-10-02T08:44:46.034395+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 11, \"unmarked\": 30}, \"B\": {\"marked\": 15, \"unmarked\": 15}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930686-dzQagndYEoqsZbhmEm0K",
        "object": "chat.completion",
        "created": 1790930686,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930686-dzQagndYEoqsZbhmEm0K",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5
      },
      "latency_s": 1.2740266249747947,
      "run": "main",
      "record_index": 199
    },
    {
      "case_id": "probability-051",
      "experiment": "probability",
      "target": 0.5,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 200,
      "started_at": "2026-10-02T08:44:47.308903+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 30,
              "unmarked": 30
            },
            "B": {
              "marked": 5,
              "unmarked": 38
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "A",
          "selection": "Choose one item uniformly at random from group A only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.52,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.52
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930687-8Z4Ja2waiFmgaeAA8JNR",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930687-8Z4Ja2waiFmgaeAA8JNR",
      "answer": {
        "type": "noul",
        "noul": 0.52
      },
      "latency_s": 0.23725399997783825,
      "run": "main",
      "record_index": 200
    },
    {
      "case_id": "probability-051",
      "experiment": "probability",
      "target": 0.5,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 201,
      "started_at": "2026-10-02T08:44:47.546582+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 30, \"unmarked\": 30}, \"B\": {\"marked\": 5, \"unmarked\": 38}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"A\", \"selection\": \"Choose one item uniformly at random from group A only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930687-PAovn3WRx5EzpaAUKcW8",
        "object": "chat.completion",
        "created": 1790930687,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930687-PAovn3WRx5EzpaAUKcW8",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5
      },
      "latency_s": 1.1557217079971451,
      "run": "main",
      "record_index": 201
    },
    {
      "case_id": "policy-011",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 202,
      "started_at": "2026-10-02T08:44:48.702884+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "orange",
            "clearance": 3,
            "emergency_permit": true,
            "experience_years": 22,
            "role": "visitor",
            "sponsor": false,
            "suspended": false,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930688-o9j8r2IVIXlyJdShAeCK",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930688-o9j8r2IVIXlyJdShAeCK",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.23826129201916046,
      "run": "main",
      "record_index": 202
    },
    {
      "case_id": "policy-011",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 203,
      "started_at": "2026-10-02T08:44:48.942222+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"orange\", \"clearance\": 3, \"emergency_permit\": true, \"experience_years\": 22, \"role\": \"visitor\", \"sponsor\": false, \"suspended\": false, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930688-uv0yny6vnx37NEQCien9",
        "object": "chat.completion",
        "created": 1790930688,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930688-uv0yny6vnx37NEQCien9",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.7294777079951018,
      "run": "main",
      "record_index": 203
    },
    {
      "case_id": "policy-010",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 204,
      "started_at": "2026-10-02T08:44:50.672405+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": false, \"days_since_purchase\": 90, \"defective\": false, \"final_sale\": true, \"price_credits\": 368, \"recalled\": false, \"receipt\": true, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930690-fnO7TM3962yPzhkxOfxy",
        "object": "chat.completion",
        "created": 1790930690,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930690-fnO7TM3962yPzhkxOfxy",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0126169580034912,
      "run": "main",
      "record_index": 204
    },
    {
      "case_id": "policy-010",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 205,
      "started_at": "2026-10-02T08:44:51.685479+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": false,
            "days_since_purchase": 90,
            "defective": false,
            "final_sale": true,
            "price_credits": 368,
            "recalled": false,
            "receipt": true,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.07,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.07
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930691-8B2LatSyYUrjObVKjHBH",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930691-8B2LatSyYUrjObVKjHBH",
      "answer": {
        "type": "noul",
        "noul": 0.07
      },
      "latency_s": 0.31941829100833274,
      "run": "main",
      "record_index": 205
    },
    {
      "case_id": "probability-043",
      "experiment": "probability",
      "target": 0.9259259259259259,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 206,
      "started_at": "2026-10-02T08:44:52.005330+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 4,
            "B": 5
          },
          "boxes": {
            "A": {
              "marked": 5,
              "unmarked": 0
            },
            "B": {
              "marked": 39,
              "unmarked": 6
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.86,
      "valid": true,
      "error": null,
      "cost_usd": 1.995e-05,
      "usage": {
        "input_tokens": 475,
        "output_tokens": 20,
        "cost": 1.995e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.86
          }
        },
        "usage": {
          "input_tokens": 475,
          "output_tokens": 20,
          "cost": 1.995e-05
        },
        "id": "gen-dec-1790930692-SdqnXS7jLCjnZ5MgRBYf",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930692-SdqnXS7jLCjnZ5MgRBYf",
      "answer": {
        "type": "noul",
        "noul": 0.86
      },
      "latency_s": 0.2204997920198366,
      "run": "main",
      "record_index": 206
    },
    {
      "case_id": "probability-043",
      "experiment": "probability",
      "target": 0.9259259259259259,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 207,
      "started_at": "2026-10-02T08:44:52.226261+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 4, \"B\": 5}, \"boxes\": {\"A\": {\"marked\": 5, \"unmarked\": 0}, \"B\": {\"marked\": 39, \"unmarked\": 6}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.92,
      "valid": true,
      "error": null,
      "cost_usd": 4.39e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 22,
        "total_tokens": 351,
        "cost": 4.39e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.39e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930692-YWHyb6JkZhYhL2A1veBA",
        "object": "chat.completion",
        "created": 1790930692,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.92}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 22,
          "total_tokens": 351,
          "cost": 4.39e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.39e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930692-YWHyb6JkZhYhL2A1veBA",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.92
      },
      "latency_s": 1.414868584019132,
      "run": "main",
      "record_index": 207
    },
    {
      "case_id": "policy-018",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 208,
      "started_at": "2026-10-02T08:44:53.641736+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": false, \"days_since_purchase\": 30, \"defective\": true, \"final_sale\": false, \"price_credits\": 298, \"recalled\": true, \"receipt\": false, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930693-27o8C83iZX60ZCUd3bL7",
        "object": "chat.completion",
        "created": 1790930693,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930693-27o8C83iZX60ZCUd3bL7",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.2540327919996344,
      "run": "main",
      "record_index": 208
    },
    {
      "case_id": "policy-018",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 209,
      "started_at": "2026-10-02T08:44:54.897024+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": false,
            "days_since_purchase": 30,
            "defective": true,
            "final_sale": false,
            "price_credits": 298,
            "recalled": true,
            "receipt": false,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930694-DMQsSWzK9pgjreTEQdxG",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930694-DMQsSWzK9pgjreTEQdxG",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.26300062498194166,
      "run": "main",
      "record_index": 209
    },
    {
      "case_id": "policy-009",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 210,
      "started_at": "2026-10-02T08:44:55.160466+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 0,
            "defective": true,
            "final_sale": true,
            "price_credits": 80,
            "recalled": false,
            "receipt": true,
            "unused": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 2.0328e-05,
      "usage": {
        "input_tokens": 484,
        "output_tokens": 20,
        "cost": 2.0328e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 484,
          "output_tokens": 20,
          "cost": 2.0328e-05
        },
        "id": "gen-dec-1790930695-hQM8UmnB9XNkaoRMWniB",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930695-hQM8UmnB9XNkaoRMWniB",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.2281371250282973,
      "run": "main",
      "record_index": 210
    },
    {
      "case_id": "policy-009",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 211,
      "started_at": "2026-10-02T08:44:55.389029+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 0, \"defective\": true, \"final_sale\": true, \"price_credits\": 80, \"recalled\": false, \"receipt\": true, \"unused\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930695-ujBesyWMWpvnLAQOBLH9",
        "object": "chat.completion",
        "created": 1790930695,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930695-ujBesyWMWpvnLAQOBLH9",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2591858340019826,
      "run": "main",
      "record_index": 211
    },
    {
      "case_id": "boolq-dev-0659",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 212,
      "started_at": "2026-10-02T08:44:56.648790+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"does the host country for the world cup get an automatic bid?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The hosts of the World Cup receive an automatic berth. Unlike many other sports, results of the previous World Cups or of the continental championships are not taken into account. Until 2002, the defending champions also received an automatic berth, but starting from the 2006 World Cup this is no longer the case.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.72e-05,
      "usage": {
        "prompt_tokens": 272,
        "completion_tokens": 20,
        "total_tokens": 292,
        "cost": 3.72e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.72e-05,
          "upstream_inference_prompt_cost": 2.72e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930696-w9dzzy0BA6IYqYrQUAsi",
        "object": "chat.completion",
        "created": 1790930696,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 272,
          "completion_tokens": 20,
          "total_tokens": 292,
          "cost": 3.72e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.72e-05,
            "upstream_inference_prompt_cost": 2.72e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930696-w9dzzy0BA6IYqYrQUAsi",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.8742578749952372,
      "run": "main",
      "record_index": 212
    },
    {
      "case_id": "boolq-dev-0659",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 213,
      "started_at": "2026-10-02T08:44:57.523488+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The hosts of the World Cup receive an automatic berth. Unlike many other sports, results of the previous World Cups or of the continental championships are not taken into account. Until 2002, the defending champions also received an automatic berth, but starting from the 2006 World Cup this is no longer the case."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "does the host country for the world cup get an automatic bid?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 1.6758e-05,
      "usage": {
        "input_tokens": 399,
        "output_tokens": 20,
        "cost": 1.6758e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 399,
          "output_tokens": 20,
          "cost": 1.6758e-05
        },
        "id": "gen-dec-1790930697-2fysqvgJ4pli1SBOb7y5",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930697-2fysqvgJ4pli1SBOb7y5",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.230788708984619,
      "run": "main",
      "record_index": 213
    },
    {
      "case_id": "probability-034",
      "experiment": "probability",
      "target": 0.98,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 214,
      "started_at": "2026-10-02T08:44:57.754749+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 49, \"unmarked\": 1}, \"B\": {\"marked\": 9, \"unmarked\": 19}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"A\", \"selection\": \"Choose one item uniformly at random from group A only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930697-Ho77ujrdYLMFAMDs1Owz",
        "object": "chat.completion",
        "created": 1790930697,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.98}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930697-Ho77ujrdYLMFAMDs1Owz",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 1.0443239580199588,
      "run": "main",
      "record_index": 214
    },
    {
      "case_id": "probability-034",
      "experiment": "probability",
      "target": 0.98,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 215,
      "started_at": "2026-10-02T08:44:58.799633+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 49,
              "unmarked": 1
            },
            "B": {
              "marked": 9,
              "unmarked": 19
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "A",
          "selection": "Choose one item uniformly at random from group A only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.96,
      "valid": true,
      "error": null,
      "cost_usd": 1.8396e-05,
      "usage": {
        "input_tokens": 438,
        "output_tokens": 20,
        "cost": 1.8396e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.96
          }
        },
        "usage": {
          "input_tokens": 438,
          "output_tokens": 20,
          "cost": 1.8396e-05
        },
        "id": "gen-dec-1790930698-nZU2kHCYu4MBZGgyA3fQ",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930698-nZU2kHCYu4MBZGgyA3fQ",
      "answer": {
        "type": "noul",
        "noul": 0.96
      },
      "latency_s": 0.23294716700911522,
      "run": "main",
      "record_index": 215
    },
    {
      "case_id": "boolq-dev-2951",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 216,
      "started_at": "2026-10-02T08:44:59.033079+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"can the word actor be used for females?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Within the profession, the re-adoption of the neutral term dates to the post-war period of the 1950 and '60s, when the contributions of women to cultural life in general were being reviewed. When The Observer and The Guardian published their new joint style guide in 2010, it stated ``Use (''actor``) for both male and female actors; do not use actress except when in name of award, e.g. Oscar for best actress.'' The guide's authors stated that ``actress comes into the same category as authoress, comedienne, manageress, 'lady doctor', 'male nurse' and similar obsolete terms that date from a time when professions were largely the preserve of one sex (usually men).'' (See male as norm). ``As Whoopi Goldberg put it in an interview with the paper: 'An actress can only play a woman. I'm an actor -- I can play anything.''' The UK performers' union Equity has no policy on the use of ``actor'' or ``actress''. An Equity spokesperson said that the union does not believe that there is a consensus on the matter and stated that the ``...subject divides the profession.'' In 2009, the Los Angeles Times stated that ``Actress'' remains the common term used in major acting awards given to female recipients (e.g., Academy Award for Best Actress).\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 5.83e-05,
      "usage": {
        "prompt_tokens": 483,
        "completion_tokens": 20,
        "total_tokens": 503,
        "cost": 5.83e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.83e-05,
          "upstream_inference_prompt_cost": 4.83e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930699-hqHzP9nnhocH5KX0ifJm",
        "object": "chat.completion",
        "created": 1790930699,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 483,
          "completion_tokens": 20,
          "total_tokens": 503,
          "cost": 5.83e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 5.83e-05,
            "upstream_inference_prompt_cost": 4.83e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930699-hqHzP9nnhocH5KX0ifJm",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1003916249901522,
      "run": "main",
      "record_index": 216
    },
    {
      "case_id": "boolq-dev-2951",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 217,
      "started_at": "2026-10-02T08:45:00.133901+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Within the profession, the re-adoption of the neutral term dates to the post-war period of the 1950 and '60s, when the contributions of women to cultural life in general were being reviewed. When The Observer and The Guardian published their new joint style guide in 2010, it stated ``Use (''actor``) for both male and female actors; do not use actress except when in name of award, e.g. Oscar for best actress.'' The guide's authors stated that ``actress comes into the same category as authoress, comedienne, manageress, 'lady doctor', 'male nurse' and similar obsolete terms that date from a time when professions were largely the preserve of one sex (usually men).'' (See male as norm). ``As Whoopi Goldberg put it in an interview with the paper: 'An actress can only play a woman. I'm an actor -- I can play anything.''' The UK performers' union Equity has no policy on the use of ``actor'' or ``actress''. An Equity spokesperson said that the union does not believe that there is a consensus on the matter and stated that the ``...subject divides the profession.'' In 2009, the Los Angeles Times stated that ``Actress'' remains the common term used in major acting awards given to female recipients (e.g., Academy Award for Best Actress)."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "can the word actor be used for females?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.97,
      "valid": true,
      "error": null,
      "cost_usd": 2.5914e-05,
      "usage": {
        "input_tokens": 617,
        "output_tokens": 20,
        "cost": 2.5914e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.97
          }
        },
        "usage": {
          "input_tokens": 617,
          "output_tokens": 20,
          "cost": 2.5914e-05
        },
        "id": "gen-dec-1790930700-LqN9bK8SZOTcVq1aFjW8",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930700-LqN9bK8SZOTcVq1aFjW8",
      "answer": {
        "type": "noul",
        "noul": 0.97
      },
      "latency_s": 0.25927891698665917,
      "run": "main",
      "record_index": 217
    },
    {
      "case_id": "probability-053",
      "experiment": "probability",
      "target": 0.49910873440285203,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 218,
      "started_at": "2026-10-02T08:45:00.393561+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 14,
            "unmarked": 20
          },
          "event": "exactly_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is exactly one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.6,
      "valid": true,
      "error": null,
      "cost_usd": 1.764e-05,
      "usage": {
        "input_tokens": 420,
        "output_tokens": 20,
        "cost": 1.764e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.6
          }
        },
        "usage": {
          "input_tokens": 420,
          "output_tokens": 20,
          "cost": 1.764e-05
        },
        "id": "gen-dec-1790930700-05oIKAfIynJYBveVcxym",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930700-05oIKAfIynJYBveVcxym",
      "answer": {
        "type": "noul",
        "noul": 0.6
      },
      "latency_s": 0.2684665840060916,
      "run": "main",
      "record_index": 218
    },
    {
      "case_id": "probability-053",
      "experiment": "probability",
      "target": 0.49910873440285203,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 219,
      "started_at": "2026-10-02T08:45:00.662424+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is exactly one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 14, \"unmarked\": 20}, \"event\": \"exactly_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5037,
      "valid": true,
      "error": null,
      "cost_usd": 3.95e-05,
      "usage": {
        "prompt_tokens": 280,
        "completion_tokens": 23,
        "total_tokens": 303,
        "cost": 3.95e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.95e-05,
          "upstream_inference_prompt_cost": 2.8e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930700-SSfpl2wWWv0qYgSeMqZF",
        "object": "chat.completion",
        "created": 1790930700,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5037}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 280,
          "completion_tokens": 23,
          "total_tokens": 303,
          "cost": 3.95e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.95e-05,
            "upstream_inference_prompt_cost": 2.8e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930700-SSfpl2wWWv0qYgSeMqZF",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5037
      },
      "latency_s": 1.3157085409911815,
      "run": "main",
      "record_index": 219
    },
    {
      "case_id": "probability-059",
      "experiment": "probability",
      "target": 0.22507836990595612,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 220,
      "started_at": "2026-10-02T08:45:01.978677+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 2,
            "B": 3
          },
          "boxes": {
            "A": {
              "marked": 1,
              "unmarked": 21
            },
            "B": {
              "marked": 10,
              "unmarked": 19
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.2,
      "valid": true,
      "error": null,
      "cost_usd": 2.0034e-05,
      "usage": {
        "input_tokens": 477,
        "output_tokens": 20,
        "cost": 2.0034e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.2
          }
        },
        "usage": {
          "input_tokens": 477,
          "output_tokens": 20,
          "cost": 2.0034e-05
        },
        "id": "gen-dec-1790930701-RG7egLJ09vlIH5Qysh6n",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930701-RG7egLJ09vlIH5Qysh6n",
      "answer": {
        "type": "noul",
        "noul": 0.2
      },
      "latency_s": 0.31081508300849237,
      "run": "main",
      "record_index": 220
    },
    {
      "case_id": "probability-059",
      "experiment": "probability",
      "target": 0.22507836990595612,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 221,
      "started_at": "2026-10-02T08:45:02.290013+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 2, \"B\": 3}, \"boxes\": {\"A\": {\"marked\": 1, \"unmarked\": 21}, \"B\": {\"marked\": 10, \"unmarked\": 19}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.2489,
      "valid": true,
      "error": null,
      "cost_usd": 4.44e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 23,
        "total_tokens": 352,
        "cost": 4.44e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.44e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930702-w3Om9ynWVrAlEpofbiHI",
        "object": "chat.completion",
        "created": 1790930702,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.2489}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 23,
          "total_tokens": 352,
          "cost": 4.44e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.44e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930702-w3Om9ynWVrAlEpofbiHI",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.2489
      },
      "latency_s": 1.3889323329785839,
      "run": "main",
      "record_index": 221
    },
    {
      "case_id": "boolq-dev-1739",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 222,
      "started_at": "2026-10-02T08:45:03.679391+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The medullary cavity (medulla, innermost part) is the central cavity of bone shafts where red bone marrow and/or yellow bone marrow (adipose tissue) is stored; hence, the medullary cavity is also known as the marrow cavity. Located in the main shaft of a long bone (diaphysis) (consisting mostly of compact bone), the medullary cavity has walls composed of spongy bone (cancellous bone) and is lined with a thin, vascular membrane (endosteum). However, the medullary cavity is the area inside any bone (long, flat, etc.) that holds the bone marrow."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is there spongy bone in the medullary cavity?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.89,
      "valid": true,
      "error": null,
      "cost_usd": 1.9698e-05,
      "usage": {
        "input_tokens": 469,
        "output_tokens": 20,
        "cost": 1.9698e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.89
          }
        },
        "usage": {
          "input_tokens": 469,
          "output_tokens": 20,
          "cost": 1.9698e-05
        },
        "id": "gen-dec-1790930703-mOxSSxS4IXbsmqAQ7LrO",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930703-mOxSSxS4IXbsmqAQ7LrO",
      "answer": {
        "type": "noul",
        "noul": 0.89
      },
      "latency_s": 0.28588129198760726,
      "run": "main",
      "record_index": 222
    },
    {
      "case_id": "boolq-dev-1739",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 223,
      "started_at": "2026-10-02T08:45:03.965996+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is there spongy bone in the medullary cavity?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The medullary cavity (medulla, innermost part) is the central cavity of bone shafts where red bone marrow and/or yellow bone marrow (adipose tissue) is stored; hence, the medullary cavity is also known as the marrow cavity. Located in the main shaft of a long bone (diaphysis) (consisting mostly of compact bone), the medullary cavity has walls composed of spongy bone (cancellous bone) and is lined with a thin, vascular membrane (endosteum). However, the medullary cavity is the area inside any bone (long, flat, etc.) that holds the bone marrow.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.44e-05,
      "usage": {
        "prompt_tokens": 344,
        "completion_tokens": 20,
        "total_tokens": 364,
        "cost": 4.44e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.44e-05,
          "upstream_inference_prompt_cost": 3.44e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930703-WDzxPk0kJOqSOLR45MY7",
        "object": "chat.completion",
        "created": 1790930703,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 344,
          "completion_tokens": 20,
          "total_tokens": 364,
          "cost": 4.44e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.44e-05,
            "upstream_inference_prompt_cost": 3.44e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930703-WDzxPk0kJOqSOLR45MY7",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.4091292500088457,
      "run": "main",
      "record_index": 223
    },
    {
      "case_id": "policy-058",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 224,
      "started_at": "2026-10-02T08:45:05.375713+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 2519, \"active_promotion\": true, \"gift_wrapping\": false, \"hazardous\": true, \"mainland\": true, \"member\": false, \"subtotal_credits\": 0, \"weight_kg\": 20}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930705-iVpfAlaNi7RbiXj0M38Z",
        "object": "chat.completion",
        "created": 1790930705,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930705-iVpfAlaNi7RbiXj0M38Z",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2133272920036688,
      "run": "main",
      "record_index": 224
    },
    {
      "case_id": "policy-058",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 225,
      "started_at": "2026-10-02T08:45:06.589513+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 2519,
            "active_promotion": true,
            "gift_wrapping": false,
            "hazardous": true,
            "mainland": true,
            "member": false,
            "subtotal_credits": 0,
            "weight_kg": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.08,
      "valid": true,
      "error": null,
      "cost_usd": 2.1252e-05,
      "usage": {
        "input_tokens": 506,
        "output_tokens": 20,
        "cost": 2.1252e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.08
          }
        },
        "usage": {
          "input_tokens": 506,
          "output_tokens": 20,
          "cost": 2.1252e-05
        },
        "id": "gen-dec-1790930706-KxUHvyo5j1PVfBe1ryZt",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930706-KxUHvyo5j1PVfBe1ryZt",
      "answer": {
        "type": "noul",
        "noul": 0.08
      },
      "latency_s": 0.2211722079955507,
      "run": "main",
      "record_index": 225
    },
    {
      "case_id": "policy-053",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 226,
      "started_at": "2026-10-02T08:45:06.811252+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 120,
            "defective": false,
            "final_sale": false,
            "price_credits": 28,
            "recalled": true,
            "receipt": true,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930706-BPmAulZSNLYAyVA0mlPI",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930706-BPmAulZSNLYAyVA0mlPI",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.331114415981574,
      "run": "main",
      "record_index": 226
    },
    {
      "case_id": "policy-053",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 227,
      "started_at": "2026-10-02T08:45:07.142754+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 120, \"defective\": false, \"final_sale\": false, \"price_credits\": 28, \"recalled\": true, \"receipt\": true, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930707-xvfeYxvr8Rfaqw2BD0QI",
        "object": "chat.completion",
        "created": 1790930707,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930707-xvfeYxvr8Rfaqw2BD0QI",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1557734580128454,
      "run": "main",
      "record_index": 227
    },
    {
      "case_id": "probability-026",
      "experiment": "probability",
      "target": 0.875,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 228,
      "started_at": "2026-10-02T08:45:08.299997+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 10,
            "unmarked": 6
          },
          "event": "at_least_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is at least one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.89,
      "valid": true,
      "error": null,
      "cost_usd": 1.764e-05,
      "usage": {
        "input_tokens": 420,
        "output_tokens": 20,
        "cost": 1.764e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.89
          }
        },
        "usage": {
          "input_tokens": 420,
          "output_tokens": 20,
          "cost": 1.764e-05
        },
        "id": "gen-dec-1790930708-LsXfaHIkKyKxmMMylrEH",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930708-LsXfaHIkKyKxmMMylrEH",
      "answer": {
        "type": "noul",
        "noul": 0.89
      },
      "latency_s": 0.2186802079959307,
      "run": "main",
      "record_index": 228
    },
    {
      "case_id": "probability-026",
      "experiment": "probability",
      "target": 0.875,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 229,
      "started_at": "2026-10-02T08:45:08.519167+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is at least one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 10, \"unmarked\": 6}, \"event\": \"at_least_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.875,
      "valid": true,
      "error": null,
      "cost_usd": 3.92e-05,
      "usage": {
        "prompt_tokens": 282,
        "completion_tokens": 22,
        "total_tokens": 304,
        "cost": 3.92e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.92e-05,
          "upstream_inference_prompt_cost": 2.82e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930708-Yg8n1XOjzNBqJMUbE54x",
        "object": "chat.completion",
        "created": 1790930708,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.875}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 282,
          "completion_tokens": 22,
          "total_tokens": 304,
          "cost": 3.92e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.92e-05,
            "upstream_inference_prompt_cost": 2.82e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930708-Yg8n1XOjzNBqJMUbE54x",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.875
      },
      "latency_s": 1.0067520839802455,
      "run": "main",
      "record_index": 229
    },
    {
      "case_id": "probability-009",
      "experiment": "probability",
      "target": 1.0,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 230,
      "started_at": "2026-10-02T08:45:09.526353+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 11,
              "unmarked": 69
            },
            "B": {
              "marked": 30,
              "unmarked": 0
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930709-ipdKKCeX84U4ANA60nFD",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930709-ipdKKCeX84U4ANA60nFD",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.26983308300259523,
      "run": "main",
      "record_index": 230
    },
    {
      "case_id": "probability-009",
      "experiment": "probability",
      "target": 1.0,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 231,
      "started_at": "2026-10-02T08:45:09.796941+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 11, \"unmarked\": 69}, \"B\": {\"marked\": 30, \"unmarked\": 0}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.91e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 20,
        "total_tokens": 311,
        "cost": 3.91e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.91e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930709-DbHSbFRf5si0Bp7jgHDS",
        "object": "chat.completion",
        "created": 1790930709,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 20,
          "total_tokens": 311,
          "cost": 3.91e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.91e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930709-DbHSbFRf5si0Bp7jgHDS",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.003813084011199,
      "run": "main",
      "record_index": 231
    },
    {
      "case_id": "boolq-dev-0308",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 232,
      "started_at": "2026-10-02T08:45:10.801438+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are ground coriander and cumin the same thing?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Coriander is commonly found both as whole dried seeds and in ground form. Roasting or heating the seeds in a dry pan heightens the flavour, aroma, and pungency. Ground coriander seed loses flavour quickly in storage and is best ground fresh. Coriander seed is a spice in garam masala and Indian curries which often employ the ground fruits in generous amounts together with cumin, acting as a thickener in a mixture called dhana jeera. Roasted coriander seeds, called dhana dal, are eaten as a snack. They are the main ingredient of the two south Indian dishes sambhar and rasam.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.32e-05,
      "usage": {
        "prompt_tokens": 332,
        "completion_tokens": 20,
        "total_tokens": 352,
        "cost": 4.32e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.32e-05,
          "upstream_inference_prompt_cost": 3.32e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930710-5c74mC26YqVpyCf6QqFQ",
        "object": "chat.completion",
        "created": 1790930710,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 332,
          "completion_tokens": 20,
          "total_tokens": 352,
          "cost": 4.32e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.32e-05,
            "upstream_inference_prompt_cost": 3.32e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930710-5c74mC26YqVpyCf6QqFQ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.181105875002686,
      "run": "main",
      "record_index": 232
    },
    {
      "case_id": "boolq-dev-0308",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 233,
      "started_at": "2026-10-02T08:45:11.982967+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Coriander is commonly found both as whole dried seeds and in ground form. Roasting or heating the seeds in a dry pan heightens the flavour, aroma, and pungency. Ground coriander seed loses flavour quickly in storage and is best ground fresh. Coriander seed is a spice in garam masala and Indian curries which often employ the ground fruits in generous amounts together with cumin, acting as a thickener in a mixture called dhana jeera. Roasted coriander seeds, called dhana dal, are eaten as a snack. They are the main ingredient of the two south Indian dishes sambhar and rasam."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are ground coriander and cumin the same thing?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.05,
      "valid": true,
      "error": null,
      "cost_usd": 1.9446e-05,
      "usage": {
        "input_tokens": 463,
        "output_tokens": 20,
        "cost": 1.9446e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.05
          }
        },
        "usage": {
          "input_tokens": 463,
          "output_tokens": 20,
          "cost": 1.9446e-05
        },
        "id": "gen-dec-1790930712-DfAd6jzjifXudWFanuSJ",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930712-DfAd6jzjifXudWFanuSJ",
      "answer": {
        "type": "noul",
        "noul": 0.05
      },
      "latency_s": 0.2550651250057854,
      "run": "main",
      "record_index": 233
    },
    {
      "case_id": "probability-050",
      "experiment": "probability",
      "target": 0.5166666666666667,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 234,
      "started_at": "2026-10-02T08:45:12.238486+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 33, \"unmarked\": 8}, \"B\": {\"marked\": 31, \"unmarked\": 29}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5166666667,
      "valid": true,
      "error": null,
      "cost_usd": 4.16e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 25,
        "total_tokens": 316,
        "cost": 4.16e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.16e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930712-MCFLoSTqaOx5tjAE8Yms",
        "object": "chat.completion",
        "created": 1790930712,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5166666667}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 25,
          "total_tokens": 316,
          "cost": 4.16e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.16e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930712-MCFLoSTqaOx5tjAE8Yms",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5166666667
      },
      "latency_s": 1.1578392080264166,
      "run": "main",
      "record_index": 234
    },
    {
      "case_id": "probability-050",
      "experiment": "probability",
      "target": 0.5166666666666667,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 235,
      "started_at": "2026-10-02T08:45:13.396782+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 33,
              "unmarked": 8
            },
            "B": {
              "marked": 31,
              "unmarked": 29
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.55,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.55
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930713-xSMfIW5EEL915xTCLf9E",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930713-xSMfIW5EEL915xTCLf9E",
      "answer": {
        "type": "noul",
        "noul": 0.55
      },
      "latency_s": 0.34863754198886454,
      "run": "main",
      "record_index": 235
    },
    {
      "case_id": "boolq-dev-1296",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 236,
      "started_at": "2026-10-02T08:45:13.745832+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "A jungle is land covered with dense vegetation dominated by trees. Application of the term has varied greatly during the past recent centuries. Before the 1970s, tropical rainforests were generally referred to as jungles but this terminology has fallen out of usage. Jungles in Western literature can represent a less civilised or unruly space outside the control of civilisation, attributed to the jungle's association in colonial discourse with places colonised by Europeans."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are the jungle and rainforest the same thing?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.39,
      "valid": true,
      "error": null,
      "cost_usd": 1.7934e-05,
      "usage": {
        "input_tokens": 427,
        "output_tokens": 20,
        "cost": 1.7934e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.39
          }
        },
        "usage": {
          "input_tokens": 427,
          "output_tokens": 20,
          "cost": 1.7934e-05
        },
        "id": "gen-dec-1790930713-20hJSBvc6bBtoF3A2MRD",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930713-20hJSBvc6bBtoF3A2MRD",
      "answer": {
        "type": "noul",
        "noul": 0.39
      },
      "latency_s": 0.3715155420068186,
      "run": "main",
      "record_index": 236
    },
    {
      "case_id": "boolq-dev-1296",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 237,
      "started_at": "2026-10-02T08:45:14.118134+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are the jungle and rainforest the same thing?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"A jungle is land covered with dense vegetation dominated by trees. Application of the term has varied greatly during the past recent centuries. Before the 1970s, tropical rainforests were generally referred to as jungles but this terminology has fallen out of usage. Jungles in Western literature can represent a less civilised or unruly space outside the control of civilisation, attributed to the jungle's association in colonial discourse with places colonised by Europeans.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.94e-05,
      "usage": {
        "prompt_tokens": 294,
        "completion_tokens": 20,
        "total_tokens": 314,
        "cost": 3.94e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.94e-05,
          "upstream_inference_prompt_cost": 2.94e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930714-evnDbOLFOjXuj6oBo0At",
        "object": "chat.completion",
        "created": 1790930714,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 294,
          "completion_tokens": 20,
          "total_tokens": 314,
          "cost": 3.94e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.94e-05,
            "upstream_inference_prompt_cost": 2.94e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930714-evnDbOLFOjXuj6oBo0At",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.1684431670000777,
      "run": "main",
      "record_index": 237
    },
    {
      "case_id": "policy-044",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 238,
      "started_at": "2026-10-02T08:45:15.287052+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 916,
            "active_promotion": true,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": true,
            "member": true,
            "subtotal_credits": 101,
            "weight_kg": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930715-2MI8UTTpNfUd5fEEBOWJ",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930715-2MI8UTTpNfUd5fEEBOWJ",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.21413104198290966,
      "run": "main",
      "record_index": 238
    },
    {
      "case_id": "policy-044",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 239,
      "started_at": "2026-10-02T08:45:15.501647+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 916, \"active_promotion\": true, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": true, \"member\": true, \"subtotal_credits\": 101, \"weight_kg\": 20}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.64e-05,
      "usage": {
        "prompt_tokens": 364,
        "completion_tokens": 20,
        "total_tokens": 384,
        "cost": 4.64e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.64e-05,
          "upstream_inference_prompt_cost": 3.64e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930715-PhORwkd3vQLDbi6jBJz9",
        "object": "chat.completion",
        "created": 1790930715,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 364,
          "completion_tokens": 20,
          "total_tokens": 384,
          "cost": 4.64e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.64e-05,
            "upstream_inference_prompt_cost": 3.64e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930715-PhORwkd3vQLDbi6jBJz9",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9860797080036718,
      "run": "main",
      "record_index": 239
    },
    {
      "case_id": "policy-035",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 240,
      "started_at": "2026-10-02T08:45:16.488227+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": false, \"days_since_purchase\": 120, \"defective\": true, \"final_sale\": true, \"price_credits\": 308, \"recalled\": false, \"receipt\": true, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930716-eAIq2v1fZh7cPPiVJ15c",
        "object": "chat.completion",
        "created": 1790930716,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930716-eAIq2v1fZh7cPPiVJ15c",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.92791120801121,
      "run": "main",
      "record_index": 240
    },
    {
      "case_id": "policy-035",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 241,
      "started_at": "2026-10-02T08:45:17.416689+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": false,
            "days_since_purchase": 120,
            "defective": true,
            "final_sale": true,
            "price_credits": 308,
            "recalled": false,
            "receipt": true,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.06,
      "valid": true,
      "error": null,
      "cost_usd": 2.0454e-05,
      "usage": {
        "input_tokens": 487,
        "output_tokens": 20,
        "cost": 2.0454e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.06
          }
        },
        "usage": {
          "input_tokens": 487,
          "output_tokens": 20,
          "cost": 2.0454e-05
        },
        "id": "gen-dec-1790930717-X0qd4H1oLNXOVjGSR7CF",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930717-X0qd4H1oLNXOVjGSR7CF",
      "answer": {
        "type": "noul",
        "noul": 0.06
      },
      "latency_s": 0.23460083399550058,
      "run": "main",
      "record_index": 241
    },
    {
      "case_id": "policy-031",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 242,
      "started_at": "2026-10-02T08:45:17.651888+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 30,
            "defective": false,
            "final_sale": false,
            "price_credits": 290,
            "recalled": true,
            "receipt": true,
            "unused": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930717-XlPNA11wJtqv5dZ6gzIr",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930717-XlPNA11wJtqv5dZ6gzIr",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.32945775002008304,
      "run": "main",
      "record_index": 242
    },
    {
      "case_id": "policy-031",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 243,
      "started_at": "2026-10-02T08:45:17.982116+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 30, \"defective\": false, \"final_sale\": false, \"price_credits\": 290, \"recalled\": true, \"receipt\": true, \"unused\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930718-Gwzz1vJIJHn8ABu3FUfO",
        "object": "chat.completion",
        "created": 1790930718,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930718-Gwzz1vJIJHn8ABu3FUfO",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.2630103749979753,
      "run": "main",
      "record_index": 243
    },
    {
      "case_id": "probability-035",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 244,
      "started_at": "2026-10-02T08:45:19.245580+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 4,
              "unmarked": 16
            },
            "B": {
              "marked": 32,
              "unmarked": 1
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "A",
          "selection": "Choose one item uniformly at random from group A only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.22,
      "valid": true,
      "error": null,
      "cost_usd": 1.8396e-05,
      "usage": {
        "input_tokens": 438,
        "output_tokens": 20,
        "cost": 1.8396e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.22
          }
        },
        "usage": {
          "input_tokens": 438,
          "output_tokens": 20,
          "cost": 1.8396e-05
        },
        "id": "gen-dec-1790930719-c1oxRPKEv3GDMqB3XiRv",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930719-c1oxRPKEv3GDMqB3XiRv",
      "answer": {
        "type": "noul",
        "noul": 0.22
      },
      "latency_s": 0.29091270800563507,
      "run": "main",
      "record_index": 244
    },
    {
      "case_id": "probability-035",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 245,
      "started_at": "2026-10-02T08:45:19.537315+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 4, \"unmarked\": 16}, \"B\": {\"marked\": 32, \"unmarked\": 1}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"A\", \"selection\": \"Choose one item uniformly at random from group A only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.2,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930719-vjgACPA3ME0ldtSOW3Vo",
        "object": "chat.completion",
        "created": 1790930719,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.2}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930719-vjgACPA3ME0ldtSOW3Vo",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.2
      },
      "latency_s": 1.2811747090017889,
      "run": "main",
      "record_index": 245
    },
    {
      "case_id": "policy-026",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 246,
      "started_at": "2026-10-02T08:45:20.818951+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "orange",
            "clearance": 1,
            "emergency_permit": true,
            "experience_years": 16,
            "role": "visitor",
            "sponsor": true,
            "suspended": false,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930720-xEzYdeZzZhWJ3pHFTuTj",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930720-xEzYdeZzZhWJ3pHFTuTj",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.24210412497632205,
      "run": "main",
      "record_index": 246
    },
    {
      "case_id": "policy-026",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 247,
      "started_at": "2026-10-02T08:45:21.061441+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"orange\", \"clearance\": 1, \"emergency_permit\": true, \"experience_years\": 16, \"role\": \"visitor\", \"sponsor\": true, \"suspended\": false, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930721-zxkUJoD0wIF3gESzpqGp",
        "object": "chat.completion",
        "created": 1790930721,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930721-zxkUJoD0wIF3gESzpqGp",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1663182500051335,
      "run": "main",
      "record_index": 247
    },
    {
      "case_id": "probability-049",
      "experiment": "probability",
      "target": 0.5772253408179631,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 248,
      "started_at": "2026-10-02T08:45:22.228272+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 7, \"B\": 3}, \"boxes\": {\"A\": {\"marked\": 19, \"unmarked\": 10}, \"B\": {\"marked\": 17, \"unmarked\": 26}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5467,
      "valid": true,
      "error": null,
      "cost_usd": 4.44e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 23,
        "total_tokens": 352,
        "cost": 4.44e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.44e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930722-FX2R1mZyJHV2csvSioce",
        "object": "chat.completion",
        "created": 1790930722,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5467}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 23,
          "total_tokens": 352,
          "cost": 4.44e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.44e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930722-FX2R1mZyJHV2csvSioce",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5467
      },
      "latency_s": 1.0084187499887776,
      "run": "main",
      "record_index": 248
    },
    {
      "case_id": "probability-049",
      "experiment": "probability",
      "target": 0.5772253408179631,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 249,
      "started_at": "2026-10-02T08:45:23.237973+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 7,
            "B": 3
          },
          "boxes": {
            "A": {
              "marked": 19,
              "unmarked": 10
            },
            "B": {
              "marked": 17,
              "unmarked": 26
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.63,
      "valid": true,
      "error": null,
      "cost_usd": 2.0076e-05,
      "usage": {
        "input_tokens": 478,
        "output_tokens": 20,
        "cost": 2.0076e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.63
          }
        },
        "usage": {
          "input_tokens": 478,
          "output_tokens": 20,
          "cost": 2.0076e-05
        },
        "id": "gen-dec-1790930723-T9z1hoLVaBFUC6QMh9ES",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930723-T9z1hoLVaBFUC6QMh9ES",
      "answer": {
        "type": "noul",
        "noul": 0.63
      },
      "latency_s": 0.23774770900490694,
      "run": "main",
      "record_index": 249
    },
    {
      "case_id": "boolq-dev-2426",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 250,
      "started_at": "2026-10-02T08:45:23.476416+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Outside the context of lawful detention or arrest, a person has no duty to answer any questions of the police. If judicial compulsion is sought by the State, the person can still invoke his or her Fifth Amendment right against compulsory self-incrimination, and refuse to testify if answers to questions posed are potentially self-incriminating. Only if granted immunity by the state, in a formal proceeding, from having any testimony or evidence derived from the testimony used against him or her, can a person be compelled to answer over an assertion of this right. If police detain (or arrest) a person, they must advise him or her that he or she has a right to remain silent, and the right to an attorney, among other rights. (This is known as the Miranda warning.) If the detained person invokes these rights, all interrogation must cease, and ordinarily nothing said by the defendant in violation of this rule may be admitted against him or her at trial."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "do you have the right to remain silent before arrest?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.205e-05,
      "usage": {
        "input_tokens": 525,
        "output_tokens": 20,
        "cost": 2.205e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 525,
          "output_tokens": 20,
          "cost": 2.205e-05
        },
        "id": "gen-dec-1790930723-hSF7IFWfDBzZpSJlVmTt",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930723-hSF7IFWfDBzZpSJlVmTt",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.23575941598392092,
      "run": "main",
      "record_index": 250
    },
    {
      "case_id": "boolq-dev-2426",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 251,
      "started_at": "2026-10-02T08:45:23.713387+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"do you have the right to remain silent before arrest?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Outside the context of lawful detention or arrest, a person has no duty to answer any questions of the police. If judicial compulsion is sought by the State, the person can still invoke his or her Fifth Amendment right against compulsory self-incrimination, and refuse to testify if answers to questions posed are potentially self-incriminating. Only if granted immunity by the state, in a formal proceeding, from having any testimony or evidence derived from the testimony used against him or her, can a person be compelled to answer over an assertion of this right. If police detain (or arrest) a person, they must advise him or her that he or she has a right to remain silent, and the right to an attorney, among other rights. (This is known as the Miranda warning.) If the detained person invokes these rights, all interrogation must cease, and ordinarily nothing said by the defendant in violation of this rule may be admitted against him or her at trial.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.98e-05,
      "usage": {
        "prompt_tokens": 398,
        "completion_tokens": 20,
        "total_tokens": 418,
        "cost": 4.98e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.98e-05,
          "upstream_inference_prompt_cost": 3.98e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930723-fBuGUl0lvh6cmAZjDEnA",
        "object": "chat.completion",
        "created": 1790930723,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 398,
          "completion_tokens": 20,
          "total_tokens": 418,
          "cost": 4.98e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.98e-05,
            "upstream_inference_prompt_cost": 3.98e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930723-fBuGUl0lvh6cmAZjDEnA",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9580867920012679,
      "run": "main",
      "record_index": 251
    },
    {
      "case_id": "probability-041",
      "experiment": "probability",
      "target": 0.0,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 252,
      "started_at": "2026-10-02T08:45:24.672280+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 0,
            "unmarked": 21
          },
          "event": "exactly_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is exactly one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.03,
      "valid": true,
      "error": null,
      "cost_usd": 1.7598e-05,
      "usage": {
        "input_tokens": 419,
        "output_tokens": 20,
        "cost": 1.7598e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.03
          }
        },
        "usage": {
          "input_tokens": 419,
          "output_tokens": 20,
          "cost": 1.7598e-05
        },
        "id": "gen-dec-1790930724-qKfmjlsrSlhU9NgKTA90",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930724-qKfmjlsrSlhU9NgKTA90",
      "answer": {
        "type": "noul",
        "noul": 0.03
      },
      "latency_s": 0.25662583400844596,
      "run": "main",
      "record_index": 252
    },
    {
      "case_id": "probability-041",
      "experiment": "probability",
      "target": 0.0,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 253,
      "started_at": "2026-10-02T08:45:24.929384+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is exactly one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 0, \"unmarked\": 21}, \"event\": \"exactly_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.8e-05,
      "usage": {
        "prompt_tokens": 280,
        "completion_tokens": 20,
        "total_tokens": 300,
        "cost": 3.8e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.8e-05,
          "upstream_inference_prompt_cost": 2.8e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930724-rOGj51K1MsPWsGsNjsii",
        "object": "chat.completion",
        "created": 1790930724,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 280,
          "completion_tokens": 20,
          "total_tokens": 300,
          "cost": 3.8e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.8e-05,
            "upstream_inference_prompt_cost": 2.8e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930724-rOGj51K1MsPWsGsNjsii",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.105800541001372,
      "run": "main",
      "record_index": 253
    },
    {
      "case_id": "policy-045",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 254,
      "started_at": "2026-10-02T08:45:26.035627+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 1675, \"active_promotion\": true, \"gift_wrapping\": true, \"hazardous\": true, \"mainland\": false, \"member\": true, \"subtotal_credits\": 50, \"weight_kg\": 21}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930726-EY8kpLurLGcBppxehG0D",
        "object": "chat.completion",
        "created": 1790930726,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930726-EY8kpLurLGcBppxehG0D",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.1904100000101607,
      "run": "main",
      "record_index": 254
    },
    {
      "case_id": "policy-045",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 255,
      "started_at": "2026-10-02T08:45:27.226487+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 1675,
            "active_promotion": true,
            "gift_wrapping": true,
            "hazardous": true,
            "mainland": false,
            "member": true,
            "subtotal_credits": 50,
            "weight_kg": 21
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.07,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.07
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930727-XwzTQKmoZnH5uFnnYWdj",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930727-XwzTQKmoZnH5uFnnYWdj",
      "answer": {
        "type": "noul",
        "noul": 0.07
      },
      "latency_s": 0.23226724998676218,
      "run": "main",
      "record_index": 255
    },
    {
      "case_id": "policy-006",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 256,
      "started_at": "2026-10-02T08:45:27.459201+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 1145, \"active_promotion\": true, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": true, \"member\": false, \"subtotal_credits\": 99, \"weight_kg\": 21}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930727-34c9hWFsJ8MlmTFIdevG",
        "object": "chat.completion",
        "created": 1790930727,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930727-34c9hWFsJ8MlmTFIdevG",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2646865420101676,
      "run": "main",
      "record_index": 256
    },
    {
      "case_id": "policy-006",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 257,
      "started_at": "2026-10-02T08:45:28.724385+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 1145,
            "active_promotion": true,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": true,
            "member": false,
            "subtotal_credits": 99,
            "weight_kg": 21
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.11,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.11
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930728-sdkPVfocHKaC2CqRQk1z",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930728-sdkPVfocHKaC2CqRQk1z",
      "answer": {
        "type": "noul",
        "noul": 0.11
      },
      "latency_s": 0.3274481660046149,
      "run": "main",
      "record_index": 257
    },
    {
      "case_id": "boolq-dev-3115",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 258,
      "started_at": "2026-10-02T08:45:29.052264+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"can i have dual citizenship in us and czech republic?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The citizenship law of the Czech Republic is based on the principles of jus sanguinis or ``right by blood''. In other words, descent from a Czech parent is the primary method of acquiring Czech citizenship (together with naturalisation). Birth on Czech territory without a Czech parent is in itself insufficient for the conferral of Czech citizenship. Every Czech citizen is also a citizen of the European Union. The law came into effect on 1 January 1993, the date of the dissolution of Czechoslovakia, and has been amended in 1993, 1995, 1996, 1999, 2002, 2003 and 2005. Since 1 January 2014, multiple citizenship under Czech law is allowed.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 4.7e-05,
      "usage": {
        "prompt_tokens": 360,
        "completion_tokens": 22,
        "total_tokens": 382,
        "cost": 4.7e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.7e-05,
          "upstream_inference_prompt_cost": 3.6e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930729-0gAVqdvo2fVIu8IHRqkZ",
        "object": "chat.completion",
        "created": 1790930729,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.98}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 360,
          "completion_tokens": 22,
          "total_tokens": 382,
          "cost": 4.7e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.7e-05,
            "upstream_inference_prompt_cost": 3.6e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930729-0gAVqdvo2fVIu8IHRqkZ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 1.1432648339832667,
      "run": "main",
      "record_index": 258
    },
    {
      "case_id": "boolq-dev-3115",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 259,
      "started_at": "2026-10-02T08:45:30.196323+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The citizenship law of the Czech Republic is based on the principles of jus sanguinis or ``right by blood''. In other words, descent from a Czech parent is the primary method of acquiring Czech citizenship (together with naturalisation). Birth on Czech territory without a Czech parent is in itself insufficient for the conferral of Czech citizenship. Every Czech citizen is also a citizen of the European Union. The law came into effect on 1 January 1993, the date of the dissolution of Czechoslovakia, and has been amended in 1993, 1995, 1996, 1999, 2002, 2003 and 2005. Since 1 January 2014, multiple citizenship under Czech law is allowed."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "can i have dual citizenship in us and czech republic?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.92,
      "valid": true,
      "error": null,
      "cost_usd": 2.1168e-05,
      "usage": {
        "input_tokens": 504,
        "output_tokens": 20,
        "cost": 2.1168e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.92
          }
        },
        "usage": {
          "input_tokens": 504,
          "output_tokens": 20,
          "cost": 2.1168e-05
        },
        "id": "gen-dec-1790930730-t0VVDrL3UC11SflcgOdj",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930730-t0VVDrL3UC11SflcgOdj",
      "answer": {
        "type": "noul",
        "noul": 0.92
      },
      "latency_s": 0.23635650001233444,
      "run": "main",
      "record_index": 259
    },
    {
      "case_id": "policy-038",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 260,
      "started_at": "2026-10-02T08:45:30.433091+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": false,
            "badge_color": "green",
            "clearance": 1,
            "emergency_permit": false,
            "experience_years": 19,
            "role": "visitor",
            "sponsor": false,
            "suspended": true,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.03,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.03
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930730-UTOXx7vNJzHr7flX08G4",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930730-UTOXx7vNJzHr7flX08G4",
      "answer": {
        "type": "noul",
        "noul": 0.03
      },
      "latency_s": 0.24581916601164266,
      "run": "main",
      "record_index": 260
    },
    {
      "case_id": "policy-038",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 261,
      "started_at": "2026-10-02T08:45:30.679380+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": false, \"badge_color\": \"green\", \"clearance\": 1, \"emergency_permit\": false, \"experience_years\": 19, \"role\": \"visitor\", \"sponsor\": false, \"suspended\": true, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930730-uZ9QpvQZXJlrrLNr9SOZ",
        "object": "chat.completion",
        "created": 1790930730,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930730-uZ9QpvQZXJlrrLNr9SOZ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.3890133750101086,
      "run": "main",
      "record_index": 261
    },
    {
      "case_id": "boolq-dev-1662",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 262,
      "started_at": "2026-10-02T08:45:32.068829+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Turf toe is named from the injury being associated with playing sports on rigid surfaces such as artificial turf and is a fairly common injury among professional American football players. Often, the injury occurs when someone or something falls on the back of the calf while that leg's knee and tips of the toes are touching the ground. The toe is hyperextended and thus the joint is injured. Additionally, athletic shoes with very flexible soles combined with cleats that ``grab'' the turf will cause overextension of the big toe. This can occur on the lesser toes as well. It has also been observed in sports beyond American football, including soccer, basketball, rugby, volleyball, and tae kwon do. This is a primary reason why many athletes prefer natural grass to turf, because it is softer."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "can you get turf toe on other toes?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.97,
      "valid": true,
      "error": null,
      "cost_usd": 2.0538e-05,
      "usage": {
        "input_tokens": 489,
        "output_tokens": 20,
        "cost": 2.0538e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.97
          }
        },
        "usage": {
          "input_tokens": 489,
          "output_tokens": 20,
          "cost": 2.0538e-05
        },
        "id": "gen-dec-1790930732-rXbPswAL5254T20dHZoE",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930732-rXbPswAL5254T20dHZoE",
      "answer": {
        "type": "noul",
        "noul": 0.97
      },
      "latency_s": 0.25860716597526334,
      "run": "main",
      "record_index": 262
    },
    {
      "case_id": "boolq-dev-1662",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 263,
      "started_at": "2026-10-02T08:45:32.327908+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"can you get turf toe on other toes?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Turf toe is named from the injury being associated with playing sports on rigid surfaces such as artificial turf and is a fairly common injury among professional American football players. Often, the injury occurs when someone or something falls on the back of the calf while that leg's knee and tips of the toes are touching the ground. The toe is hyperextended and thus the joint is injured. Additionally, athletic shoes with very flexible soles combined with cleats that ``grab'' the turf will cause overextension of the big toe. This can occur on the lesser toes as well. It has also been observed in sports beyond American football, including soccer, basketball, rugby, volleyball, and tae kwon do. This is a primary reason why many athletes prefer natural grass to turf, because it is softer.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.62e-05,
      "usage": {
        "prompt_tokens": 362,
        "completion_tokens": 20,
        "total_tokens": 382,
        "cost": 4.62e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.62e-05,
          "upstream_inference_prompt_cost": 3.62e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930732-K67K1N36qjZo9eyQfFHq",
        "object": "chat.completion",
        "created": 1790930732,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 362,
          "completion_tokens": 20,
          "total_tokens": 382,
          "cost": 4.62e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.62e-05,
            "upstream_inference_prompt_cost": 3.62e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930732-K67K1N36qjZo9eyQfFHq",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0418905839906074,
      "run": "main",
      "record_index": 263
    },
    {
      "case_id": "policy-003",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 264,
      "started_at": "2026-10-02T08:45:33.370312+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 113, \"active_promotion\": false, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": false, \"member\": true, \"subtotal_credits\": 99, \"weight_kg\": 21}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.64e-05,
      "usage": {
        "prompt_tokens": 364,
        "completion_tokens": 20,
        "total_tokens": 384,
        "cost": 4.64e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.64e-05,
          "upstream_inference_prompt_cost": 3.64e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930733-cvt4oFGGkfHx0rMCupJX",
        "object": "chat.completion",
        "created": 1790930733,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 364,
          "completion_tokens": 20,
          "total_tokens": 384,
          "cost": 4.64e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.64e-05,
            "upstream_inference_prompt_cost": 3.64e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930733-cvt4oFGGkfHx0rMCupJX",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.256898874999024,
      "run": "main",
      "record_index": 264
    },
    {
      "case_id": "policy-003",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 265,
      "started_at": "2026-10-02T08:45:34.627672+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 113,
            "active_promotion": false,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": false,
            "member": true,
            "subtotal_credits": 99,
            "weight_kg": 21
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.05,
      "valid": true,
      "error": null,
      "cost_usd": 2.1252e-05,
      "usage": {
        "input_tokens": 506,
        "output_tokens": 20,
        "cost": 2.1252e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.05
          }
        },
        "usage": {
          "input_tokens": 506,
          "output_tokens": 20,
          "cost": 2.1252e-05
        },
        "id": "gen-dec-1790930734-xN3dqA0Lt1KHdG0vyFcm",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930734-xN3dqA0Lt1KHdG0vyFcm",
      "answer": {
        "type": "noul",
        "noul": 0.05
      },
      "latency_s": 0.22727533400757238,
      "run": "main",
      "record_index": 265
    },
    {
      "case_id": "boolq-dev-2597",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 266,
      "started_at": "2026-10-02T08:45:34.855437+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is hellmann's and best foods the same company?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Hellmann's and Best Foods are brand names that are used for the same line of mayonnaise and other food products. The Hellmann's brand is sold in the United States east of the Rocky Mountains, and also in Latin America, Europe, Australia, the Middle East, Canada and South Africa. The Best Foods brand is sold in the United States west of the Rocky Mountains, and also in Asia, Australia, and New Zealand.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.25,
      "valid": true,
      "error": null,
      "cost_usd": 4.03e-05,
      "usage": {
        "prompt_tokens": 293,
        "completion_tokens": 22,
        "total_tokens": 315,
        "cost": 4.03e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.03e-05,
          "upstream_inference_prompt_cost": 2.93e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930734-6LC7Nzf5evOcteM1XSCe",
        "object": "chat.completion",
        "created": 1790930734,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.25}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 293,
          "completion_tokens": 22,
          "total_tokens": 315,
          "cost": 4.03e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.03e-05,
            "upstream_inference_prompt_cost": 2.93e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930734-6LC7Nzf5evOcteM1XSCe",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.25
      },
      "latency_s": 0.9698960829991847,
      "run": "main",
      "record_index": 266
    },
    {
      "case_id": "boolq-dev-2597",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 267,
      "started_at": "2026-10-02T08:45:35.825795+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Hellmann's and Best Foods are brand names that are used for the same line of mayonnaise and other food products. The Hellmann's brand is sold in the United States east of the Rocky Mountains, and also in Latin America, Europe, Australia, the Middle East, Canada and South Africa. The Best Foods brand is sold in the United States west of the Rocky Mountains, and also in Asia, Australia, and New Zealand."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is hellmann's and best foods the same company?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.96,
      "valid": true,
      "error": null,
      "cost_usd": 1.7556e-05,
      "usage": {
        "input_tokens": 418,
        "output_tokens": 20,
        "cost": 1.7556e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.96
          }
        },
        "usage": {
          "input_tokens": 418,
          "output_tokens": 20,
          "cost": 1.7556e-05
        },
        "id": "gen-dec-1790930735-Q3Dp9q6Pew2K1DUe0sOd",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930735-Q3Dp9q6Pew2K1DUe0sOd",
      "answer": {
        "type": "noul",
        "noul": 0.96
      },
      "latency_s": 0.2237324159941636,
      "run": "main",
      "record_index": 267
    },
    {
      "case_id": "probability-002",
      "experiment": "probability",
      "target": 0.7863247863247863,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 268,
      "started_at": "2026-10-02T08:45:36.049925+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 24,
            "unmarked": 3
          },
          "event": "both_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Are both selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.77,
      "valid": true,
      "error": null,
      "cost_usd": 1.7304e-05,
      "usage": {
        "input_tokens": 412,
        "output_tokens": 20,
        "cost": 1.7304e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.77
          }
        },
        "usage": {
          "input_tokens": 412,
          "output_tokens": 20,
          "cost": 1.7304e-05
        },
        "id": "gen-dec-1790930736-Akc9Ne2Zw0deTN9S0DTP",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930736-Akc9Ne2Zw0deTN9S0DTP",
      "answer": {
        "type": "noul",
        "noul": 0.77
      },
      "latency_s": 0.23464312497526407,
      "run": "main",
      "record_index": 268
    },
    {
      "case_id": "probability-002",
      "experiment": "probability",
      "target": 0.7863247863247863,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 269,
      "started_at": "2026-10-02T08:45:36.284998+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Are both selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 24, \"unmarked\": 3}, \"event\": \"both_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.819404,
      "valid": true,
      "error": null,
      "cost_usd": 3.89e-05,
      "usage": {
        "prompt_tokens": 274,
        "completion_tokens": 23,
        "total_tokens": 297,
        "cost": 3.89e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.89e-05,
          "upstream_inference_prompt_cost": 2.74e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930736-7mPCPOJY0WFtDdOQHIen",
        "object": "chat.completion",
        "created": 1790930736,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.819404}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 274,
          "completion_tokens": 23,
          "total_tokens": 297,
          "cost": 3.89e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.89e-05,
            "upstream_inference_prompt_cost": 2.74e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930736-7mPCPOJY0WFtDdOQHIen",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.819404
      },
      "latency_s": 1.2336802089994308,
      "run": "main",
      "record_index": 269
    },
    {
      "case_id": "policy-059",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 270,
      "started_at": "2026-10-02T08:45:37.519189+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"orange\", \"clearance\": 3, \"emergency_permit\": true, \"experience_years\": 13, \"role\": \"contractor\", \"sponsor\": false, \"suspended\": false, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.62e-05,
      "usage": {
        "prompt_tokens": 362,
        "completion_tokens": 20,
        "total_tokens": 382,
        "cost": 4.62e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.62e-05,
          "upstream_inference_prompt_cost": 3.62e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930737-PDC30eEvwnqctAy5eYHB",
        "object": "chat.completion",
        "created": 1790930737,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 362,
          "completion_tokens": 20,
          "total_tokens": 382,
          "cost": 4.62e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.62e-05,
            "upstream_inference_prompt_cost": 3.62e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930737-PDC30eEvwnqctAy5eYHB",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.8998147910169791,
      "run": "main",
      "record_index": 270
    },
    {
      "case_id": "policy-059",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 271,
      "started_at": "2026-10-02T08:45:38.419382+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "orange",
            "clearance": 3,
            "emergency_permit": true,
            "experience_years": 13,
            "role": "contractor",
            "sponsor": false,
            "suspended": false,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1126e-05,
      "usage": {
        "input_tokens": 503,
        "output_tokens": 20,
        "cost": 2.1126e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 503,
          "output_tokens": 20,
          "cost": 2.1126e-05
        },
        "id": "gen-dec-1790930738-oqhPfnQXGwnp64X4CvML",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930738-oqhPfnQXGwnp64X4CvML",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.39210662501864135,
      "run": "main",
      "record_index": 271
    },
    {
      "case_id": "probability-029",
      "experiment": "probability",
      "target": 0.8894230769230769,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 272,
      "started_at": "2026-10-02T08:45:38.811855+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 7,
            "B": 6
          },
          "boxes": {
            "A": {
              "marked": 45,
              "unmarked": 3
            },
            "B": {
              "marked": 30,
              "unmarked": 6
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.87,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.87
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930738-i0XDS6VX2UAkKhBco5cq",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930738-i0XDS6VX2UAkKhBco5cq",
      "answer": {
        "type": "noul",
        "noul": 0.87
      },
      "latency_s": 0.23217737500090152,
      "run": "main",
      "record_index": 272
    },
    {
      "case_id": "probability-029",
      "experiment": "probability",
      "target": 0.8894230769230769,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 273,
      "started_at": "2026-10-02T08:45:39.044369+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 7, \"B\": 6}, \"boxes\": {\"A\": {\"marked\": 45, \"unmarked\": 3}, \"B\": {\"marked\": 30, \"unmarked\": 6}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.925,
      "valid": true,
      "error": null,
      "cost_usd": 4.39e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 22,
        "total_tokens": 351,
        "cost": 4.39e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.39e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930739-Wa6KNqiwHkTSVOnUhOYP",
        "object": "chat.completion",
        "created": 1790930739,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.925}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 22,
          "total_tokens": 351,
          "cost": 4.39e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.39e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930739-Wa6KNqiwHkTSVOnUhOYP",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.925
      },
      "latency_s": 1.337724957993487,
      "run": "main",
      "record_index": 273
    },
    {
      "case_id": "policy-050",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 274,
      "started_at": "2026-10-02T08:45:40.382532+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 1795, \"active_promotion\": false, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": true, \"member\": true, \"subtotal_credits\": 101, \"weight_kg\": 20}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930740-Ur4SvcKB1g5v95tXEGcH",
        "object": "chat.completion",
        "created": 1790930740,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930740-Ur4SvcKB1g5v95tXEGcH",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.2039998330001254,
      "run": "main",
      "record_index": 274
    },
    {
      "case_id": "policy-050",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 275,
      "started_at": "2026-10-02T08:45:41.586972+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 1795,
            "active_promotion": false,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": true,
            "member": true,
            "subtotal_credits": 101,
            "weight_kg": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1336e-05,
      "usage": {
        "input_tokens": 508,
        "output_tokens": 20,
        "cost": 2.1336e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 508,
          "output_tokens": 20,
          "cost": 2.1336e-05
        },
        "id": "gen-dec-1790930741-NZl2WUzlEbg5TTNjssBd",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930741-NZl2WUzlEbg5TTNjssBd",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.3065278750145808,
      "run": "main",
      "record_index": 275
    },
    {
      "case_id": "boolq-dev-1381",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 276,
      "started_at": "2026-10-02T08:45:41.893876+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"can a catcher block home plate without the ball?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"In baseball, blocking the plate is a technique performed by a catcher to prevent a runner from scoring. The act of blocking the plate accounted for most of the physical contact in Major League Baseball prior to the 2014 season, when it was outlawed except when the catcher already has possession of the ball.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.67e-05,
      "usage": {
        "prompt_tokens": 267,
        "completion_tokens": 20,
        "total_tokens": 287,
        "cost": 3.67e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.67e-05,
          "upstream_inference_prompt_cost": 2.67e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930741-XTZ7oPn8DfwSaZAKUaI3",
        "object": "chat.completion",
        "created": 1790930741,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 267,
          "completion_tokens": 20,
          "total_tokens": 287,
          "cost": 3.67e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.67e-05,
            "upstream_inference_prompt_cost": 2.67e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930741-XTZ7oPn8DfwSaZAKUaI3",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.166173333011102,
      "run": "main",
      "record_index": 276
    },
    {
      "case_id": "boolq-dev-1381",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 277,
      "started_at": "2026-10-02T08:45:43.060445+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "In baseball, blocking the plate is a technique performed by a catcher to prevent a runner from scoring. The act of blocking the plate accounted for most of the physical contact in Major League Baseball prior to the 2014 season, when it was outlawed except when the catcher already has possession of the ball."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "can a catcher block home plate without the ball?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.15,
      "valid": true,
      "error": null,
      "cost_usd": 1.6506e-05,
      "usage": {
        "input_tokens": 393,
        "output_tokens": 20,
        "cost": 1.6506e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.15
          }
        },
        "usage": {
          "input_tokens": 393,
          "output_tokens": 20,
          "cost": 1.6506e-05
        },
        "id": "gen-dec-1790930743-8MhS4HrrnLtwPqDxt1Pc",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930743-8MhS4HrrnLtwPqDxt1Pc",
      "answer": {
        "type": "noul",
        "noul": 0.15
      },
      "latency_s": 0.31710983300581574,
      "run": "main",
      "record_index": 277
    },
    {
      "case_id": "policy-048",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 278,
      "started_at": "2026-10-02T08:45:43.377889+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": false,
            "badge_color": "blue",
            "clearance": 3,
            "emergency_permit": true,
            "experience_years": 25,
            "role": "staff",
            "sponsor": false,
            "suspended": true,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.13,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.13
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930743-ChJ35X59dicH5vfe5otx",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930743-ChJ35X59dicH5vfe5otx",
      "answer": {
        "type": "noul",
        "noul": 0.13
      },
      "latency_s": 0.2654526660044212,
      "run": "main",
      "record_index": 278
    },
    {
      "case_id": "policy-048",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 279,
      "started_at": "2026-10-02T08:45:43.643716+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": false, \"badge_color\": \"blue\", \"clearance\": 3, \"emergency_permit\": true, \"experience_years\": 25, \"role\": \"staff\", \"sponsor\": false, \"suspended\": true, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930743-Oc7i6nAQvewr1gp68X4h",
        "object": "chat.completion",
        "created": 1790930743,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930743-Oc7i6nAQvewr1gp68X4h",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2671437920071185,
      "run": "main",
      "record_index": 279
    },
    {
      "case_id": "boolq-dev-0210",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 280,
      "started_at": "2026-10-02T08:45:44.911723+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"did murphy brown have a baby on the show?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"In the show's 1991--92 season, Murphy became pregnant. When her baby's father (ex-husband and current underground radical Jake Lowenstein) expressed his unwillingness to give up his own lifestyle to be a parent, Murphy chose to have the child and raise it alone. Another major fiction-reality blending came at Murphy's baby shower: the invited guests were journalists Katie Couric, Joan Lunden, Paula Zahn, Mary Alice Williams and Faith Daniels, who treated the fictional Murphy and Corky as friends and peers.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.13e-05,
      "usage": {
        "prompt_tokens": 313,
        "completion_tokens": 20,
        "total_tokens": 333,
        "cost": 4.13e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.13e-05,
          "upstream_inference_prompt_cost": 3.13e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930744-369FNE8Jxx5S2udbR9VL",
        "object": "chat.completion",
        "created": 1790930744,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 313,
          "completion_tokens": 20,
          "total_tokens": 333,
          "cost": 4.13e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.13e-05,
            "upstream_inference_prompt_cost": 3.13e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930744-369FNE8Jxx5S2udbR9VL",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.230047083023237,
      "run": "main",
      "record_index": 280
    },
    {
      "case_id": "boolq-dev-0210",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 281,
      "started_at": "2026-10-02T08:45:46.142369+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "In the show's 1991--92 season, Murphy became pregnant. When her baby's father (ex-husband and current underground radical Jake Lowenstein) expressed his unwillingness to give up his own lifestyle to be a parent, Murphy chose to have the child and raise it alone. Another major fiction-reality blending came at Murphy's baby shower: the invited guests were journalists Katie Couric, Joan Lunden, Paula Zahn, Mary Alice Williams and Faith Daniels, who treated the fictional Murphy and Corky as friends and peers."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "did murphy brown have a baby on the show?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.96,
      "valid": true,
      "error": null,
      "cost_usd": 1.869e-05,
      "usage": {
        "input_tokens": 445,
        "output_tokens": 20,
        "cost": 1.869e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.96
          }
        },
        "usage": {
          "input_tokens": 445,
          "output_tokens": 20,
          "cost": 1.869e-05
        },
        "id": "gen-dec-1790930746-8OoavaQU4weE7y3RqTZy",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930746-8OoavaQU4weE7y3RqTZy",
      "answer": {
        "type": "noul",
        "noul": 0.96
      },
      "latency_s": 0.352098625007784,
      "run": "main",
      "record_index": 281
    },
    {
      "case_id": "boolq-dev-2849",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 282,
      "started_at": "2026-10-02T08:45:46.494790+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Although it is widely believed that a worker honey bee can sting only once, this is a partial misconception: although the stinger is in fact barbed so that it lodges in the victim's skin, tearing loose from the bee's abdomen and leading to its death in minutes, this only happens if the skin of the victim is sufficiently thick, such as a mammal's. Honey bees are the only hymenoptera with a strongly barbed sting, though yellow jackets and some other wasps have small barbs."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is there always a stinger in a bee sting?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.46,
      "valid": true,
      "error": null,
      "cost_usd": 1.827e-05,
      "usage": {
        "input_tokens": 435,
        "output_tokens": 20,
        "cost": 1.827e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.46
          }
        },
        "usage": {
          "input_tokens": 435,
          "output_tokens": 20,
          "cost": 1.827e-05
        },
        "id": "gen-dec-1790930746-WyqNxJ7OOxlbFFkHoKUE",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930746-WyqNxJ7OOxlbFFkHoKUE",
      "answer": {
        "type": "noul",
        "noul": 0.46
      },
      "latency_s": 0.22556845800136216,
      "run": "main",
      "record_index": 282
    },
    {
      "case_id": "boolq-dev-2849",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 283,
      "started_at": "2026-10-02T08:45:46.720906+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is there always a stinger in a bee sting?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Although it is widely believed that a worker honey bee can sting only once, this is a partial misconception: although the stinger is in fact barbed so that it lodges in the victim's skin, tearing loose from the bee's abdomen and leading to its death in minutes, this only happens if the skin of the victim is sufficiently thick, such as a mammal's. Honey bees are the only hymenoptera with a strongly barbed sting, though yellow jackets and some other wasps have small barbs.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.05,
      "valid": true,
      "error": null,
      "cost_usd": 4.22e-05,
      "usage": {
        "prompt_tokens": 312,
        "completion_tokens": 22,
        "total_tokens": 334,
        "cost": 4.22e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.22e-05,
          "upstream_inference_prompt_cost": 3.12e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930746-yTf9lc2gXmBYxP5SEh4t",
        "object": "chat.completion",
        "created": 1790930746,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.05}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 312,
          "completion_tokens": 22,
          "total_tokens": 334,
          "cost": 4.22e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.22e-05,
            "upstream_inference_prompt_cost": 3.12e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930746-yTf9lc2gXmBYxP5SEh4t",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.05
      },
      "latency_s": 1.267879791004816,
      "run": "main",
      "record_index": 283
    },
    {
      "case_id": "probability-006",
      "experiment": "probability",
      "target": 0.28,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 284,
      "started_at": "2026-10-02T08:45:47.989623+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is exactly one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 4, \"unmarked\": 21}, \"event\": \"exactly_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.56,
      "valid": true,
      "error": null,
      "cost_usd": 3.9e-05,
      "usage": {
        "prompt_tokens": 280,
        "completion_tokens": 22,
        "total_tokens": 302,
        "cost": 3.9e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.9e-05,
          "upstream_inference_prompt_cost": 2.8e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930748-xEFAten88UGwPOikxHEp",
        "object": "chat.completion",
        "created": 1790930748,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.56}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 280,
          "completion_tokens": 22,
          "total_tokens": 302,
          "cost": 3.9e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.9e-05,
            "upstream_inference_prompt_cost": 2.8e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930748-xEFAten88UGwPOikxHEp",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.56
      },
      "latency_s": 0.9234876669943333,
      "run": "main",
      "record_index": 284
    },
    {
      "case_id": "probability-006",
      "experiment": "probability",
      "target": 0.28,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 285,
      "started_at": "2026-10-02T08:45:48.913520+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 4,
            "unmarked": 21
          },
          "event": "exactly_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is exactly one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.41,
      "valid": true,
      "error": null,
      "cost_usd": 1.7598e-05,
      "usage": {
        "input_tokens": 419,
        "output_tokens": 20,
        "cost": 1.7598e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.41
          }
        },
        "usage": {
          "input_tokens": 419,
          "output_tokens": 20,
          "cost": 1.7598e-05
        },
        "id": "gen-dec-1790930748-ZdsQm5per20UHIQecUor",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930748-ZdsQm5per20UHIQecUor",
      "answer": {
        "type": "noul",
        "noul": 0.41
      },
      "latency_s": 0.27165437501389533,
      "run": "main",
      "record_index": 285
    },
    {
      "case_id": "boolq-dev-2774",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 286,
      "started_at": "2026-10-02T08:45:49.185653+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are walgreens and rite aid owned by the same company?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Rite Aid began in 1962, opening its first store in Scranton, Pennsylvania; it was called Thrift D Discount Center. After several years of growth, Rite Aid adopted its current name and debuted as a public company in 1968. As of 2017, Rite Aid is publicly traded on the New York Stock Exchange under the symbol RAD. Its major competitors are CVS and Walgreens. In late 2015, Walgreens announced that it would acquire Rite Aid for $9.4 billion pending approval. However, on June 29, 2017, over fear of antitrust regulations, Walgreens Boots Alliance announced it would buy roughly half of Rite Aid's stores for $5.18 billion. On September 19, 2017, the Federal Trade Commission (FTC) approved a fourth deal agreement to purchase Rite Aid with 1,932 stores for $4.38 billion total.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.05,
      "valid": true,
      "error": null,
      "cost_usd": 5e-05,
      "usage": {
        "prompt_tokens": 390,
        "completion_tokens": 22,
        "total_tokens": 412,
        "cost": 5e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5e-05,
          "upstream_inference_prompt_cost": 3.9e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930749-Wn0V0qHm4R2ZiToJLulj",
        "object": "chat.completion",
        "created": 1790930749,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.05}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 390,
          "completion_tokens": 22,
          "total_tokens": 412,
          "cost": 5e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 5e-05,
            "upstream_inference_prompt_cost": 3.9e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930749-Wn0V0qHm4R2ZiToJLulj",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.05
      },
      "latency_s": 1.1853979169973172,
      "run": "main",
      "record_index": 286
    },
    {
      "case_id": "boolq-dev-2774",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 287,
      "started_at": "2026-10-02T08:45:50.371444+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Rite Aid began in 1962, opening its first store in Scranton, Pennsylvania; it was called Thrift D Discount Center. After several years of growth, Rite Aid adopted its current name and debuted as a public company in 1968. As of 2017, Rite Aid is publicly traded on the New York Stock Exchange under the symbol RAD. Its major competitors are CVS and Walgreens. In late 2015, Walgreens announced that it would acquire Rite Aid for $9.4 billion pending approval. However, on June 29, 2017, over fear of antitrust regulations, Walgreens Boots Alliance announced it would buy roughly half of Rite Aid's stores for $5.18 billion. On September 19, 2017, the Federal Trade Commission (FTC) approved a fourth deal agreement to purchase Rite Aid with 1,932 stores for $4.38 billion total."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are walgreens and rite aid owned by the same company?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.39,
      "valid": true,
      "error": null,
      "cost_usd": 2.2722e-05,
      "usage": {
        "input_tokens": 541,
        "output_tokens": 20,
        "cost": 2.2722e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.39
          }
        },
        "usage": {
          "input_tokens": 541,
          "output_tokens": 20,
          "cost": 2.2722e-05
        },
        "id": "gen-dec-1790930750-HDkfxCxtSD9pc3VV5H2h",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930750-HDkfxCxtSD9pc3VV5H2h",
      "answer": {
        "type": "noul",
        "noul": 0.39
      },
      "latency_s": 0.22156704199733213,
      "run": "main",
      "record_index": 287
    },
    {
      "case_id": "probability-004",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 288,
      "started_at": "2026-10-02T08:45:50.593497+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 7,
              "unmarked": 9
            },
            "B": {
              "marked": 4,
              "unmarked": 16
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.22,
      "valid": true,
      "error": null,
      "cost_usd": 1.8354e-05,
      "usage": {
        "input_tokens": 437,
        "output_tokens": 20,
        "cost": 1.8354e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.22
          }
        },
        "usage": {
          "input_tokens": 437,
          "output_tokens": 20,
          "cost": 1.8354e-05
        },
        "id": "gen-dec-1790930750-4fkcv8RYYIuYvLlP0Nig",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930750-4fkcv8RYYIuYvLlP0Nig",
      "answer": {
        "type": "noul",
        "noul": 0.22
      },
      "latency_s": 0.25701229099649936,
      "run": "main",
      "record_index": 288
    },
    {
      "case_id": "probability-004",
      "experiment": "probability",
      "target": 0.2,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 289,
      "started_at": "2026-10-02T08:45:50.851007+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 7, \"unmarked\": 9}, \"B\": {\"marked\": 4, \"unmarked\": 16}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.2,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930750-5LVcVEHA2uNvEwilSojD",
        "object": "chat.completion",
        "created": 1790930750,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.2}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930750-5LVcVEHA2uNvEwilSojD",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.2
      },
      "latency_s": 1.562041708995821,
      "run": "main",
      "record_index": 289
    },
    {
      "case_id": "boolq-dev-0852",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 290,
      "started_at": "2026-10-02T08:45:52.413464+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"can a girl scout be an eagle scout?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The Gold Award is often compared to the Eagle Scout in the Boy Scouts of America.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5,
      "valid": true,
      "error": null,
      "cost_usd": 3.32e-05,
      "usage": {
        "prompt_tokens": 222,
        "completion_tokens": 22,
        "total_tokens": 244,
        "cost": 3.32e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.32e-05,
          "upstream_inference_prompt_cost": 2.22e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930752-Nbd8UGCI3rRIMngd2oJs",
        "object": "chat.completion",
        "created": 1790930752,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 222,
          "completion_tokens": 22,
          "total_tokens": 244,
          "cost": 3.32e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.32e-05,
            "upstream_inference_prompt_cost": 2.22e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930752-Nbd8UGCI3rRIMngd2oJs",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5
      },
      "latency_s": 0.9779684169916436,
      "run": "main",
      "record_index": 290
    },
    {
      "case_id": "boolq-dev-0852",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 291,
      "started_at": "2026-10-02T08:45:53.391809+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The Gold Award is often compared to the Eagle Scout in the Boy Scouts of America."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "can a girl scout be an eagle scout?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.11,
      "valid": true,
      "error": null,
      "cost_usd": 1.449e-05,
      "usage": {
        "input_tokens": 345,
        "output_tokens": 20,
        "cost": 1.449e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.11
          }
        },
        "usage": {
          "input_tokens": 345,
          "output_tokens": 20,
          "cost": 1.449e-05
        },
        "id": "gen-dec-1790930753-XVpUmnqU9dmmkUPo5JLM",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930753-XVpUmnqU9dmmkUPo5JLM",
      "answer": {
        "type": "noul",
        "noul": 0.11
      },
      "latency_s": 0.575849584012758,
      "run": "main",
      "record_index": 291
    },
    {
      "case_id": "boolq-dev-2868",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 292,
      "started_at": "2026-10-02T08:45:53.968240+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Open carry is also legal throughout North Carolina. In the town of Chapel Hill, open carry is restricted to guns of a certain minimum size, under the theory that small, concealable handguns are more often associated with criminal activity. No permit is required to carry a handgun openly in North Carolina. In the court case of State v. Kerner(1921) the defendant ended up getting into some type of confrontation with another man. The defendant proceeded to walk back to his place of work, get his gun, and then return to the scene to fight. The defendant ended up being charged with ``carrying a concealed weapon'' and ``carrying his pistol off his premises unconcealed,'' which violated a local act applicable to Forsyth County and ended up being a misdemeanor. The defendant was taken to trial and the trial judge then dismissed the charge as unconstitutional. The state then appealed, and the supreme court affirmed. During court, the court stated at the beginning that the Second Amendment did not apply, because ``the first ten amendments to the United States Constitution are restrictions on the federal authority and not the states.'' Therefore, with that being said, it focused more on the state constitution. The state constitution states that: ``A well regulated militia, being necessary to the security of a free state, the right of the people to keep and bear arms, shall not be infringed.'' The court viewed the provision as protecting the right to carry arms in public. Forsyth County's local act was condemned and seen as distasteful, because it ended up putting a restriction on a persons right to carry a pistol, more so an unconcealed pistol. Although, the case of State v. Kerner helped/made more clear the allowance of openly carrying a pistol, it does not preclude all regulations regarding the carrying of firearms."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "do you need a permit to open carry in nc?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.07,
      "valid": true,
      "error": null,
      "cost_usd": 2.9568e-05,
      "usage": {
        "input_tokens": 704,
        "output_tokens": 20,
        "cost": 2.9568e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.07
          }
        },
        "usage": {
          "input_tokens": 704,
          "output_tokens": 20,
          "cost": 2.9568e-05
        },
        "id": "gen-dec-1790930753-twtDPvmnS69lBAI5tP0g",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930753-twtDPvmnS69lBAI5tP0g",
      "answer": {
        "type": "noul",
        "noul": 0.07
      },
      "latency_s": 0.22563112500938587,
      "run": "main",
      "record_index": 292
    },
    {
      "case_id": "boolq-dev-2868",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 293,
      "started_at": "2026-10-02T08:45:54.194209+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"do you need a permit to open carry in nc?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Open carry is also legal throughout North Carolina. In the town of Chapel Hill, open carry is restricted to guns of a certain minimum size, under the theory that small, concealable handguns are more often associated with criminal activity. No permit is required to carry a handgun openly in North Carolina. In the court case of State v. Kerner(1921) the defendant ended up getting into some type of confrontation with another man. The defendant proceeded to walk back to his place of work, get his gun, and then return to the scene to fight. The defendant ended up being charged with ``carrying a concealed weapon'' and ``carrying his pistol off his premises unconcealed,'' which violated a local act applicable to Forsyth County and ended up being a misdemeanor. The defendant was taken to trial and the trial judge then dismissed the charge as unconstitutional. The state then appealed, and the supreme court affirmed. During court, the court stated at the beginning that the Second Amendment did not apply, because ``the first ten amendments to the United States Constitution are restrictions on the federal authority and not the states.'' Therefore, with that being said, it focused more on the state constitution. The state constitution states that: ``A well regulated militia, being necessary to the security of a free state, the right of the people to keep and bear arms, shall not be infringed.'' The court viewed the provision as protecting the right to carry arms in public. Forsyth County's local act was condemned and seen as distasteful, because it ended up putting a restriction on a persons right to carry a pistol, more so an unconcealed pistol. Although, the case of State v. Kerner helped/made more clear the allowance of openly carrying a pistol, it does not preclude all regulations regarding the carrying of firearms.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 6.74e-05,
      "usage": {
        "prompt_tokens": 574,
        "completion_tokens": 20,
        "total_tokens": 594,
        "cost": 6.74e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 6.74e-05,
          "upstream_inference_prompt_cost": 5.74e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930754-IXoHoT90K52IBv8Jpy89",
        "object": "chat.completion",
        "created": 1790930754,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 574,
          "completion_tokens": 20,
          "total_tokens": 594,
          "cost": 6.74e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 6.74e-05,
            "upstream_inference_prompt_cost": 5.74e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930754-IXoHoT90K52IBv8Jpy89",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1917702500068117,
      "run": "main",
      "record_index": 293
    },
    {
      "case_id": "boolq-dev-3247",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 294,
      "started_at": "2026-10-02T08:45:55.386614+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The fifth season of the television comedy series Arrested Development premiered on Netflix on May 29, 2018. The season will consist of 16 episodes, split into two eight-episode parts; with the second half premiering later in 2018. This is the second revival season after the series was canceled by Fox in 2006; the fourth season premiered in 2013."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is there a season 5 of arrested development?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 1.7724e-05,
      "usage": {
        "input_tokens": 422,
        "output_tokens": 20,
        "cost": 1.7724e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 422,
          "output_tokens": 20,
          "cost": 1.7724e-05
        },
        "id": "gen-dec-1790930755-GAmq1wreZKDWxl4eWhsm",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930755-GAmq1wreZKDWxl4eWhsm",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.23406683400389738,
      "run": "main",
      "record_index": 294
    },
    {
      "case_id": "boolq-dev-3247",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 295,
      "started_at": "2026-10-02T08:45:55.621286+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is there a season 5 of arrested development?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The fifth season of the television comedy series Arrested Development premiered on Netflix on May 29, 2018. The season will consist of 16 episodes, split into two eight-episode parts; with the second half premiering later in 2018. This is the second revival season after the series was canceled by Fox in 2006; the fourth season premiered in 2013.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.85e-05,
      "usage": {
        "prompt_tokens": 285,
        "completion_tokens": 20,
        "total_tokens": 305,
        "cost": 3.85e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.85e-05,
          "upstream_inference_prompt_cost": 2.85e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930755-4jZzVKoJpvbs6cJBZVtn",
        "object": "chat.completion",
        "created": 1790930755,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 285,
          "completion_tokens": 20,
          "total_tokens": 305,
          "cost": 3.85e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.85e-05,
            "upstream_inference_prompt_cost": 2.85e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930755-4jZzVKoJpvbs6cJBZVtn",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1535711660108063,
      "run": "main",
      "record_index": 295
    },
    {
      "case_id": "boolq-dev-1603",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 296,
      "started_at": "2026-10-02T08:45:56.775320+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Father's Day is a celebration honoring fathers and celebrating fatherhood, paternal bonds, and the influence of fathers in society. In Catholic Europe, it has been celebrated on March 19 (St. Joseph's Day) since the Middle Ages. This celebration was brought by the Spanish and Portuguese to Latin America, where March 19 is often still used for it, though many countries in Europe and the Americas have adopted the U.S. date, which is the third Sunday of June. It is celebrated on various days in many parts of the world, most commonly in the months of March, April and June. It complements similar celebrations honoring family members, such as Mother's Day, Siblings Day, and Grandparents' Day."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is father's day the same day all over the world?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.04,
      "valid": true,
      "error": null,
      "cost_usd": 2.0286e-05,
      "usage": {
        "input_tokens": 483,
        "output_tokens": 20,
        "cost": 2.0286e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.04
          }
        },
        "usage": {
          "input_tokens": 483,
          "output_tokens": 20,
          "cost": 2.0286e-05
        },
        "id": "gen-dec-1790930756-0TeQBDHwL1OoEtwvaCi7",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930756-0TeQBDHwL1OoEtwvaCi7",
      "answer": {
        "type": "noul",
        "noul": 0.04
      },
      "latency_s": 0.2674277079931926,
      "run": "main",
      "record_index": 296
    },
    {
      "case_id": "boolq-dev-1603",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 297,
      "started_at": "2026-10-02T08:45:57.043149+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is father's day the same day all over the world?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Father's Day is a celebration honoring fathers and celebrating fatherhood, paternal bonds, and the influence of fathers in society. In Catholic Europe, it has been celebrated on March 19 (St. Joseph's Day) since the Middle Ages. This celebration was brought by the Spanish and Portuguese to Latin America, where March 19 is often still used for it, though many countries in Europe and the Americas have adopted the U.S. date, which is the third Sunday of June. It is celebrated on various days in many parts of the world, most commonly in the months of March, April and June. It complements similar celebrations honoring family members, such as Mother's Day, Siblings Day, and Grandparents' Day.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.51e-05,
      "usage": {
        "prompt_tokens": 351,
        "completion_tokens": 20,
        "total_tokens": 371,
        "cost": 4.51e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.51e-05,
          "upstream_inference_prompt_cost": 3.51e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930757-ql8UGriNqfjQn6uFDgds",
        "object": "chat.completion",
        "created": 1790930757,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 351,
          "completion_tokens": 20,
          "total_tokens": 371,
          "cost": 4.51e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.51e-05,
            "upstream_inference_prompt_cost": 3.51e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930757-ql8UGriNqfjQn6uFDgds",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0467605410085525,
      "run": "main",
      "record_index": 297
    },
    {
      "case_id": "probability-058",
      "experiment": "probability",
      "target": 0.6691086691086691,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 298,
      "started_at": "2026-10-02T08:45:58.090375+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 8,
            "B": 1
          },
          "boxes": {
            "A": {
              "marked": 30,
              "unmarked": 12
            },
            "B": {
              "marked": 4,
              "unmarked": 9
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.69,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.69
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930758-miBBJctbCKUUbXsjbDM0",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930758-miBBJctbCKUUbXsjbDM0",
      "answer": {
        "type": "noul",
        "noul": 0.69
      },
      "latency_s": 0.24065591700491495,
      "run": "main",
      "record_index": 298
    },
    {
      "case_id": "probability-058",
      "experiment": "probability",
      "target": 0.6691086691086691,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 299,
      "started_at": "2026-10-02T08:45:58.331436+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 8, \"B\": 1}, \"boxes\": {\"A\": {\"marked\": 30, \"unmarked\": 12}, \"B\": {\"marked\": 4, \"unmarked\": 9}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.7989,
      "valid": true,
      "error": null,
      "cost_usd": 4.44e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 23,
        "total_tokens": 352,
        "cost": 4.44e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.44e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930758-ajCLebeO6zdB1KfgP0rg",
        "object": "chat.completion",
        "created": 1790930758,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.7989}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 23,
          "total_tokens": 352,
          "cost": 4.44e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.44e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930758-ajCLebeO6zdB1KfgP0rg",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.7989
      },
      "latency_s": 1.143257709016325,
      "run": "main",
      "record_index": 299
    },
    {
      "case_id": "policy-014",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 300,
      "started_at": "2026-10-02T08:45:59.475268+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 90, \"defective\": true, \"final_sale\": true, \"price_credits\": 141, \"recalled\": false, \"receipt\": false, \"unused\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930759-7AHmJq9EocEjau5L5KMt",
        "object": "chat.completion",
        "created": 1790930759,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930759-7AHmJq9EocEjau5L5KMt",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.9255749589938205,
      "run": "main",
      "record_index": 300
    },
    {
      "case_id": "policy-014",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 301,
      "started_at": "2026-10-02T08:46:01.401385+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 90,
            "defective": true,
            "final_sale": true,
            "price_credits": 141,
            "recalled": false,
            "receipt": false,
            "unused": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930761-FiNKQNLYLeBdqjVNRj7e",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930761-FiNKQNLYLeBdqjVNRj7e",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.2441956659895368,
      "run": "main",
      "record_index": 301
    },
    {
      "case_id": "policy-020",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 302,
      "started_at": "2026-10-02T08:46:01.645996+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "green",
            "clearance": 3,
            "emergency_permit": true,
            "experience_years": 17,
            "role": "visitor",
            "sponsor": false,
            "suspended": true,
            "training_complete": true
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930761-j4j3P0Gjb0t3xR18iHR8",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930761-j4j3P0Gjb0t3xR18iHR8",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.2340466669993475,
      "run": "main",
      "record_index": 302
    },
    {
      "case_id": "policy-020",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 303,
      "started_at": "2026-10-02T08:46:01.880497+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"green\", \"clearance\": 3, \"emergency_permit\": true, \"experience_years\": 17, \"role\": \"visitor\", \"sponsor\": false, \"suspended\": true, \"training_complete\": true}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930761-koWeKXDHCcUpWGYGGumz",
        "object": "chat.completion",
        "created": 1790930761,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930761-koWeKXDHCcUpWGYGGumz",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2097005840041675,
      "run": "main",
      "record_index": 303
    },
    {
      "case_id": "boolq-dev-2116",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 304,
      "started_at": "2026-10-02T08:46:03.090697+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is father's day celebrated in the netherlands?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"In the Netherlands, Father's Day (Vaderdag) is celebrated on the third Sunday of June and is not a public holiday. Traditionally, as on Mother's Day, fathers get breakfast in bed made by their children and families gather together and have dinner, usually at the grandparents' house. In recent years, families also started having dinner out, and as on Mother's Day, it is one of the busiest days for restaurants. At school, children handcraft their present for their fathers. Consumer goods companies have all sorts of special offers for fathers: socks, ties, electronics, suits, and men's healthcare products.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.26e-05,
      "usage": {
        "prompt_tokens": 326,
        "completion_tokens": 20,
        "total_tokens": 346,
        "cost": 4.26e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.26e-05,
          "upstream_inference_prompt_cost": 3.26e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930763-xodPjiIiSTZK6h5qy1fN",
        "object": "chat.completion",
        "created": 1790930763,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 326,
          "completion_tokens": 20,
          "total_tokens": 346,
          "cost": 4.26e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.26e-05,
            "upstream_inference_prompt_cost": 3.26e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930763-xodPjiIiSTZK6h5qy1fN",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.3064640829979908,
      "run": "main",
      "record_index": 304
    },
    {
      "case_id": "boolq-dev-2116",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 305,
      "started_at": "2026-10-02T08:46:04.397607+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "In the Netherlands, Father's Day (Vaderdag) is celebrated on the third Sunday of June and is not a public holiday. Traditionally, as on Mother's Day, fathers get breakfast in bed made by their children and families gather together and have dinner, usually at the grandparents' house. In recent years, families also started having dinner out, and as on Mother's Day, it is one of the busiest days for restaurants. At school, children handcraft their present for their fathers. Consumer goods companies have all sorts of special offers for fathers: socks, ties, electronics, suits, and men's healthcare products."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is father's day celebrated in the netherlands?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 1.9278e-05,
      "usage": {
        "input_tokens": 459,
        "output_tokens": 20,
        "cost": 1.9278e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 459,
          "output_tokens": 20,
          "cost": 1.9278e-05
        },
        "id": "gen-dec-1790930764-kbcsrE6Hq0gtsvhVBZgG",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930764-kbcsrE6Hq0gtsvhVBZgG",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.22123799999826588,
      "run": "main",
      "record_index": 305
    },
    {
      "case_id": "policy-033",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 306,
      "started_at": "2026-10-02T08:46:04.619255+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 2604,
            "active_promotion": true,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": true,
            "member": false,
            "subtotal_credits": 101,
            "weight_kg": 21
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.13,
      "valid": true,
      "error": null,
      "cost_usd": 2.1336e-05,
      "usage": {
        "input_tokens": 508,
        "output_tokens": 20,
        "cost": 2.1336e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.13
          }
        },
        "usage": {
          "input_tokens": 508,
          "output_tokens": 20,
          "cost": 2.1336e-05
        },
        "id": "gen-dec-1790930764-kPwZphC93v0VPfSnoeLK",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930764-kPwZphC93v0VPfSnoeLK",
      "answer": {
        "type": "noul",
        "noul": 0.13
      },
      "latency_s": 0.24405379098607227,
      "run": "main",
      "record_index": 306
    },
    {
      "case_id": "policy-033",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 307,
      "started_at": "2026-10-02T08:46:04.863817+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 2604, \"active_promotion\": true, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": true, \"member\": false, \"subtotal_credits\": 101, \"weight_kg\": 21}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930764-oTw5GWMkFDxRozp5jPyp",
        "object": "chat.completion",
        "created": 1790930764,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930764-oTw5GWMkFDxRozp5jPyp",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2135736670170445,
      "run": "main",
      "record_index": 307
    },
    {
      "case_id": "policy-029",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 308,
      "started_at": "2026-10-02T08:46:06.078394+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 2564, \"active_promotion\": true, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": false, \"member\": true, \"subtotal_credits\": 99, \"weight_kg\": 19}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930766-8lfxXQ5VUubJGPCnzDQU",
        "object": "chat.completion",
        "created": 1790930766,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930766-8lfxXQ5VUubJGPCnzDQU",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.8982982079905923,
      "run": "main",
      "record_index": 308
    },
    {
      "case_id": "policy-029",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 309,
      "started_at": "2026-10-02T08:46:06.977137+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 2564,
            "active_promotion": true,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": false,
            "member": true,
            "subtotal_credits": 99,
            "weight_kg": 19
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930766-dXo4HckyLzkfw4B0P08X",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930766-dXo4HckyLzkfw4B0P08X",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.22041695800726302,
      "run": "main",
      "record_index": 309
    },
    {
      "case_id": "policy-055",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 310,
      "started_at": "2026-10-02T08:46:07.198754+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 2316,
            "active_promotion": true,
            "gift_wrapping": true,
            "hazardous": false,
            "mainland": true,
            "member": false,
            "subtotal_credits": 49,
            "weight_kg": 19
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.97,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.97
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930767-fByRe3G4aliBV6VGmWOG",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930767-fByRe3G4aliBV6VGmWOG",
      "answer": {
        "type": "noul",
        "noul": 0.97
      },
      "latency_s": 0.22052295901812613,
      "run": "main",
      "record_index": 310
    },
    {
      "case_id": "policy-055",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 311,
      "started_at": "2026-10-02T08:46:07.419674+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 2316, \"active_promotion\": true, \"gift_wrapping\": true, \"hazardous\": false, \"mainland\": true, \"member\": false, \"subtotal_credits\": 49, \"weight_kg\": 19}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930767-0DpYxC7yAGTjRs1Y4iUN",
        "object": "chat.completion",
        "created": 1790930767,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930767-0DpYxC7yAGTjRs1Y4iUN",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.243754249997437,
      "run": "main",
      "record_index": 311
    },
    {
      "case_id": "policy-005",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 312,
      "started_at": "2026-10-02T08:46:08.664694+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": false,
            "days_since_purchase": 90,
            "defective": true,
            "final_sale": false,
            "price_credits": 22,
            "recalled": true,
            "receipt": false,
            "unused": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.037e-05,
      "usage": {
        "input_tokens": 485,
        "output_tokens": 20,
        "cost": 2.037e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 485,
          "output_tokens": 20,
          "cost": 2.037e-05
        },
        "id": "gen-dec-1790930768-uGiSNQewclgDfnIYrkwN",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930768-uGiSNQewclgDfnIYrkwN",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.23372087499592453,
      "run": "main",
      "record_index": 312
    },
    {
      "case_id": "policy-005",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 313,
      "started_at": "2026-10-02T08:46:08.898859+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": false, \"days_since_purchase\": 90, \"defective\": true, \"final_sale\": false, \"price_credits\": 22, \"recalled\": true, \"receipt\": false, \"unused\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930768-R8dd1unjAGoNtpji6ru2",
        "object": "chat.completion",
        "created": 1790930768,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930768-R8dd1unjAGoNtpji6ru2",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1990753340069205,
      "run": "main",
      "record_index": 313
    },
    {
      "case_id": "policy-030",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 314,
      "started_at": "2026-10-02T08:46:10.098411+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "green",
            "clearance": 3,
            "emergency_permit": true,
            "experience_years": 25,
            "role": "contractor",
            "sponsor": true,
            "suspended": true,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.07,
      "valid": true,
      "error": null,
      "cost_usd": 2.1126e-05,
      "usage": {
        "input_tokens": 503,
        "output_tokens": 20,
        "cost": 2.1126e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.07
          }
        },
        "usage": {
          "input_tokens": 503,
          "output_tokens": 20,
          "cost": 2.1126e-05
        },
        "id": "gen-dec-1790930770-mdu2yNDuYoxNGoiCBazk",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930770-mdu2yNDuYoxNGoiCBazk",
      "answer": {
        "type": "noul",
        "noul": 0.07
      },
      "latency_s": 0.23757541700615548,
      "run": "main",
      "record_index": 314
    },
    {
      "case_id": "policy-030",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 315,
      "started_at": "2026-10-02T08:46:10.336425+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"green\", \"clearance\": 3, \"emergency_permit\": true, \"experience_years\": 25, \"role\": \"contractor\", \"sponsor\": true, \"suspended\": true, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.62e-05,
      "usage": {
        "prompt_tokens": 362,
        "completion_tokens": 20,
        "total_tokens": 382,
        "cost": 4.62e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.62e-05,
          "upstream_inference_prompt_cost": 3.62e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930770-p9G6T1w5k7a4lWpEmeZ4",
        "object": "chat.completion",
        "created": 1790930770,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 362,
          "completion_tokens": 20,
          "total_tokens": 382,
          "cost": 4.62e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.62e-05,
            "upstream_inference_prompt_cost": 3.62e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930770-p9G6T1w5k7a4lWpEmeZ4",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0476681659929454,
      "run": "main",
      "record_index": 315
    },
    {
      "case_id": "probability-012",
      "experiment": "probability",
      "target": 0.8890469416785206,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 316,
      "started_at": "2026-10-02T08:46:11.384703+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 25,
            "unmarked": 13
          },
          "event": "at_least_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is at least one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.86,
      "valid": true,
      "error": null,
      "cost_usd": 1.7682e-05,
      "usage": {
        "input_tokens": 421,
        "output_tokens": 20,
        "cost": 1.7682e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.86
          }
        },
        "usage": {
          "input_tokens": 421,
          "output_tokens": 20,
          "cost": 1.7682e-05
        },
        "id": "gen-dec-1790930771-xOh2Fd3Q3nVeFFPXLNok",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930771-xOh2Fd3Q3nVeFFPXLNok",
      "answer": {
        "type": "noul",
        "noul": 0.86
      },
      "latency_s": 0.22357020899653435,
      "run": "main",
      "record_index": 316
    },
    {
      "case_id": "probability-012",
      "experiment": "probability",
      "target": 0.8890469416785206,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 317,
      "started_at": "2026-10-02T08:46:11.609551+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is at least one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 25, \"unmarked\": 13}, \"event\": \"at_least_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.8654,
      "valid": true,
      "error": null,
      "cost_usd": 3.97e-05,
      "usage": {
        "prompt_tokens": 282,
        "completion_tokens": 23,
        "total_tokens": 305,
        "cost": 3.97e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.97e-05,
          "upstream_inference_prompt_cost": 2.82e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930771-iHTDP8A0tMW4GJOHtqaJ",
        "object": "chat.completion",
        "created": 1790930771,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.8654}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 282,
          "completion_tokens": 23,
          "total_tokens": 305,
          "cost": 3.97e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.97e-05,
            "upstream_inference_prompt_cost": 2.82e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930771-iHTDP8A0tMW4GJOHtqaJ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.8654
      },
      "latency_s": 1.3331656249938533,
      "run": "main",
      "record_index": 317
    },
    {
      "case_id": "policy-022",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 318,
      "started_at": "2026-10-02T08:46:12.943206+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"orange\", \"clearance\": 1, \"emergency_permit\": true, \"experience_years\": 29, \"role\": \"staff\", \"sponsor\": false, \"suspended\": false, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.61e-05,
      "usage": {
        "prompt_tokens": 361,
        "completion_tokens": 20,
        "total_tokens": 381,
        "cost": 4.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.61e-05,
          "upstream_inference_prompt_cost": 3.61e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930772-a6xlOxRIiua4y5tJ0r6Y",
        "object": "chat.completion",
        "created": 1790930772,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 361,
          "completion_tokens": 20,
          "total_tokens": 381,
          "cost": 4.61e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.61e-05,
            "upstream_inference_prompt_cost": 3.61e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930772-a6xlOxRIiua4y5tJ0r6Y",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.072620624996489,
      "run": "main",
      "record_index": 318
    },
    {
      "case_id": "policy-022",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 319,
      "started_at": "2026-10-02T08:46:14.016319+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "orange",
            "clearance": 1,
            "emergency_permit": true,
            "experience_years": 29,
            "role": "staff",
            "sponsor": false,
            "suspended": false,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930774-Ax0ox7GwgXhG0zAvgDTS",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930774-Ax0ox7GwgXhG0zAvgDTS",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.2193983329925686,
      "run": "main",
      "record_index": 319
    },
    {
      "case_id": "probability-013",
      "experiment": "probability",
      "target": 0.95,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 320,
      "started_at": "2026-10-02T08:46:14.236141+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 51, \"unmarked\": 11}, \"B\": {\"marked\": 19, \"unmarked\": 1}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.95,
      "valid": true,
      "error": null,
      "cost_usd": 4.01e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 22,
        "total_tokens": 313,
        "cost": 4.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.01e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930774-tdV5nhuJnUkeSpkHsVa4",
        "object": "chat.completion",
        "created": 1790930774,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.95}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 22,
          "total_tokens": 313,
          "cost": 4.01e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.01e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930774-tdV5nhuJnUkeSpkHsVa4",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.95
      },
      "latency_s": 1.1419865839998238,
      "run": "main",
      "record_index": 320
    },
    {
      "case_id": "probability-013",
      "experiment": "probability",
      "target": 0.95,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 321,
      "started_at": "2026-10-02T08:46:15.378849+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 51,
              "unmarked": 11
            },
            "B": {
              "marked": 19,
              "unmarked": 1
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.94,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.94
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930775-oJhXJQBkzMW2aq0vJH5Y",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930775-oJhXJQBkzMW2aq0vJH5Y",
      "answer": {
        "type": "noul",
        "noul": 0.94
      },
      "latency_s": 0.23400391600443982,
      "run": "main",
      "record_index": 321
    },
    {
      "case_id": "boolq-dev-3134",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 322,
      "started_at": "2026-10-02T08:46:15.613358+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"can a person be a citizen of no country?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"In International law a stateless person is someone who is ``not considered as a national by any state under the operation of its law''. Some stateless persons are also refugees. However, not all refugees are stateless, and many persons who are stateless have never crossed an international border.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.64e-05,
      "usage": {
        "prompt_tokens": 264,
        "completion_tokens": 20,
        "total_tokens": 284,
        "cost": 3.64e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.64e-05,
          "upstream_inference_prompt_cost": 2.64e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930775-OTMevJxwdbuIURARRjnf",
        "object": "chat.completion",
        "created": 1790930775,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 264,
          "completion_tokens": 20,
          "total_tokens": 284,
          "cost": 3.64e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.64e-05,
            "upstream_inference_prompt_cost": 2.64e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930775-OTMevJxwdbuIURARRjnf",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.5631497089925688,
      "run": "main",
      "record_index": 322
    },
    {
      "case_id": "boolq-dev-3134",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 323,
      "started_at": "2026-10-02T08:46:17.177231+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "In International law a stateless person is someone who is ``not considered as a national by any state under the operation of its law''. Some stateless persons are also refugees. However, not all refugees are stateless, and many persons who are stateless have never crossed an international border."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "can a person be a citizen of no country?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 1.6254e-05,
      "usage": {
        "input_tokens": 387,
        "output_tokens": 20,
        "cost": 1.6254e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 387,
          "output_tokens": 20,
          "cost": 1.6254e-05
        },
        "id": "gen-dec-1790930777-a9LJ82ZaLho6ytXGaBV5",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930777-a9LJ82ZaLho6ytXGaBV5",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.2279814169742167,
      "run": "main",
      "record_index": 323
    },
    {
      "case_id": "boolq-dev-1953",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 324,
      "started_at": "2026-10-02T08:46:17.405605+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is all of australia in the same time zone?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Australia uses three main time zones: Australian Western Standard Time (AWST; UTC+08:00), Australian Central Standard Time (ACST; UTC+09:30), and Australian Eastern Standard Time (AEST; UTC+10:00). Time is regulated by the individual state governments, some of which observe daylight saving time (DST). Australia's external territories observe different time zones.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.83e-05,
      "usage": {
        "prompt_tokens": 283,
        "completion_tokens": 20,
        "total_tokens": 303,
        "cost": 3.83e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.83e-05,
          "upstream_inference_prompt_cost": 2.83e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930777-oBgM1RMgndzQWgOsXk1Q",
        "object": "chat.completion",
        "created": 1790930777,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 283,
          "completion_tokens": 20,
          "total_tokens": 303,
          "cost": 3.83e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.83e-05,
            "upstream_inference_prompt_cost": 2.83e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930777-oBgM1RMgndzQWgOsXk1Q",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.469710709003266,
      "run": "main",
      "record_index": 324
    },
    {
      "case_id": "boolq-dev-1953",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 325,
      "started_at": "2026-10-02T08:46:18.875762+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Australia uses three main time zones: Australian Western Standard Time (AWST; UTC+08:00), Australian Central Standard Time (ACST; UTC+09:30), and Australian Eastern Standard Time (AEST; UTC+10:00). Time is regulated by the individual state governments, some of which observe daylight saving time (DST). Australia's external territories observe different time zones."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is all of australia in the same time zone?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.02,
      "valid": true,
      "error": null,
      "cost_usd": 1.7346e-05,
      "usage": {
        "input_tokens": 413,
        "output_tokens": 20,
        "cost": 1.7346e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.02
          }
        },
        "usage": {
          "input_tokens": 413,
          "output_tokens": 20,
          "cost": 1.7346e-05
        },
        "id": "gen-dec-1790930778-f6u0jPEjIAdfaVGaeubT",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930778-f6u0jPEjIAdfaVGaeubT",
      "answer": {
        "type": "noul",
        "noul": 0.02
      },
      "latency_s": 0.27875029199640267,
      "run": "main",
      "record_index": 325
    },
    {
      "case_id": "policy-054",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 326,
      "started_at": "2026-10-02T08:46:19.154940+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 2681, \"active_promotion\": false, \"gift_wrapping\": false, \"hazardous\": false, \"mainland\": true, \"member\": true, \"subtotal_credits\": 51, \"weight_kg\": 19}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930779-JRbRiddtmUWfOGKXhsMj",
        "object": "chat.completion",
        "created": 1790930779,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930779-JRbRiddtmUWfOGKXhsMj",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0271641250001267,
      "run": "main",
      "record_index": 326
    },
    {
      "case_id": "policy-054",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 327,
      "started_at": "2026-10-02T08:46:20.182546+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 2681,
            "active_promotion": false,
            "gift_wrapping": false,
            "hazardous": false,
            "mainland": true,
            "member": true,
            "subtotal_credits": 51,
            "weight_kg": 19
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930780-hEOuqWzfcdUVGFQuIMzd",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930780-hEOuqWzfcdUVGFQuIMzd",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.22947650001151487,
      "run": "main",
      "record_index": 327
    },
    {
      "case_id": "boolq-dev-2818",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 328,
      "started_at": "2026-10-02T08:46:20.412449+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Agronomists describe the benefits to yield in rotated crops as ``The Rotation Effect''. There are many found benefits of rotation systems: however, there is no specific scientific basis for the sometimes 10-25% yield increase in a crop grown in rotation versus monoculture. The factors related to the increase are simply described as alleviation of the negative factors of monoculture cropping systems. Explanations due to improved nutrition; pest, pathogen, and weed stress reduction; and improved soil structure have been found in some cases to be correlated, but causation has not been determined for the majority of cropping systems."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is there a way one can grow more crops from the same land?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.88,
      "valid": true,
      "error": null,
      "cost_usd": 1.9362e-05,
      "usage": {
        "input_tokens": 461,
        "output_tokens": 20,
        "cost": 1.9362e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.88
          }
        },
        "usage": {
          "input_tokens": 461,
          "output_tokens": 20,
          "cost": 1.9362e-05
        },
        "id": "gen-dec-1790930780-YXK82aFdFJPuyiXRpMxn",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930780-YXK82aFdFJPuyiXRpMxn",
      "answer": {
        "type": "noul",
        "noul": 0.88
      },
      "latency_s": 0.26372412499040365,
      "run": "main",
      "record_index": 328
    },
    {
      "case_id": "boolq-dev-2818",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 329,
      "started_at": "2026-10-02T08:46:20.676885+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is there a way one can grow more crops from the same land?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Agronomists describe the benefits to yield in rotated crops as ``The Rotation Effect''. There are many found benefits of rotation systems: however, there is no specific scientific basis for the sometimes 10-25% yield increase in a crop grown in rotation versus monoculture. The factors related to the increase are simply described as alleviation of the negative factors of monoculture cropping systems. Explanations due to improved nutrition; pest, pathogen, and weed stress reduction; and improved soil structure have been found in some cases to be correlated, but causation has not been determined for the majority of cropping systems.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.5,
      "valid": true,
      "error": null,
      "cost_usd": 4.41e-05,
      "usage": {
        "prompt_tokens": 331,
        "completion_tokens": 22,
        "total_tokens": 353,
        "cost": 4.41e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.41e-05,
          "upstream_inference_prompt_cost": 3.31e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930780-2B5BN5d4ki8eVYOmqOdp",
        "object": "chat.completion",
        "created": 1790930780,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.5}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 331,
          "completion_tokens": 22,
          "total_tokens": 353,
          "cost": 4.41e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.41e-05,
            "upstream_inference_prompt_cost": 3.31e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930780-2B5BN5d4ki8eVYOmqOdp",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.5
      },
      "latency_s": 1.0206769589858595,
      "run": "main",
      "record_index": 329
    },
    {
      "case_id": "policy-004",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 330,
      "started_at": "2026-10-02T08:46:21.698103+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": true, \"days_since_purchase\": 30, \"defective\": false, \"final_sale\": true, \"price_credits\": 256, \"recalled\": false, \"receipt\": false, \"unused\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930781-UhDxDp9McCN0FdSE9e09",
        "object": "chat.completion",
        "created": 1790930781,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930781-UhDxDp9McCN0FdSE9e09",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.642270958982408,
      "run": "main",
      "record_index": 330
    },
    {
      "case_id": "policy-004",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 331,
      "started_at": "2026-10-02T08:46:23.340857+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": true,
            "days_since_purchase": 30,
            "defective": false,
            "final_sale": true,
            "price_credits": 256,
            "recalled": false,
            "receipt": false,
            "unused": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.07,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.07
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930783-xBi8xnGsKpLHOQyIRDKA",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930783-xBi8xnGsKpLHOQyIRDKA",
      "answer": {
        "type": "noul",
        "noul": 0.07
      },
      "latency_s": 0.23038941601407714,
      "run": "main",
      "record_index": 331
    },
    {
      "case_id": "boolq-dev-0383",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 332,
      "started_at": "2026-10-02T08:46:23.571726+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The hosts of the World Cup receive an automatic berth. Unlike many other sports, results of the previous World Cups or of the continental championships are not taken into account. Until 2002, the defending champions also received an automatic berth, but starting from the 2006 World Cup this is no longer the case."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "do you qualify for the world cup if you host it?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 1.6716e-05,
      "usage": {
        "input_tokens": 398,
        "output_tokens": 20,
        "cost": 1.6716e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 398,
          "output_tokens": 20,
          "cost": 1.6716e-05
        },
        "id": "gen-dec-1790930783-aXLU32lZbozHu3qAJ0sP",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930783-aXLU32lZbozHu3qAJ0sP",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.22430791697115637,
      "run": "main",
      "record_index": 332
    },
    {
      "case_id": "boolq-dev-0383",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 333,
      "started_at": "2026-10-02T08:46:23.796484+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"do you qualify for the world cup if you host it?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The hosts of the World Cup receive an automatic berth. Unlike many other sports, results of the previous World Cups or of the continental championships are not taken into account. Until 2002, the defending champions also received an automatic berth, but starting from the 2006 World Cup this is no longer the case.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.71e-05,
      "usage": {
        "prompt_tokens": 271,
        "completion_tokens": 20,
        "total_tokens": 291,
        "cost": 3.71e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.71e-05,
          "upstream_inference_prompt_cost": 2.71e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930783-QoUZ7d3JtO87zE4Xhaq1",
        "object": "chat.completion",
        "created": 1790930783,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 271,
          "completion_tokens": 20,
          "total_tokens": 291,
          "cost": 3.71e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.71e-05,
            "upstream_inference_prompt_cost": 2.71e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930783-QoUZ7d3JtO87zE4Xhaq1",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1194969580101315,
      "run": "main",
      "record_index": 333
    },
    {
      "case_id": "probability-037",
      "experiment": "probability",
      "target": 0.03333333333333333,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 334,
      "started_at": "2026-10-02T08:46:24.916849+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 21, \"unmarked\": 27}, \"B\": {\"marked\": 1, \"unmarked\": 29}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.03333333333333333,
      "valid": true,
      "error": null,
      "cost_usd": 4.26e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 27,
        "total_tokens": 318,
        "cost": 4.26e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.26e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1.35e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930784-sMAE0sndDCzp6mzEnRYA",
        "object": "chat.completion",
        "created": 1790930784,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.03333333333333333}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 27,
          "total_tokens": 318,
          "cost": 4.26e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.26e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1.35e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930784-sMAE0sndDCzp6mzEnRYA",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.03333333333333333
      },
      "latency_s": 1.49891504200059,
      "run": "main",
      "record_index": 334
    },
    {
      "case_id": "probability-037",
      "experiment": "probability",
      "target": 0.03333333333333333,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 335,
      "started_at": "2026-10-02T08:46:26.416254+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 21,
              "unmarked": 27
            },
            "B": {
              "marked": 1,
              "unmarked": 29
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.06,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.06
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930786-7m6fc7vPtkb7o4mnSClH",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930786-7m6fc7vPtkb7o4mnSClH",
      "answer": {
        "type": "noul",
        "noul": 0.06
      },
      "latency_s": 0.2342511249880772,
      "run": "main",
      "record_index": 335
    },
    {
      "case_id": "probability-036",
      "experiment": "probability",
      "target": 0.10476190476190476,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 336,
      "started_at": "2026-10-02T08:46:26.651090+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 12,
            "unmarked": 24
          },
          "event": "both_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Are both selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.28,
      "valid": true,
      "error": null,
      "cost_usd": 1.7346e-05,
      "usage": {
        "input_tokens": 413,
        "output_tokens": 20,
        "cost": 1.7346e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.28
          }
        },
        "usage": {
          "input_tokens": 413,
          "output_tokens": 20,
          "cost": 1.7346e-05
        },
        "id": "gen-dec-1790930786-Woj7VrYk6u0wiUdn7R5T",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930786-Woj7VrYk6u0wiUdn7R5T",
      "answer": {
        "type": "noul",
        "noul": 0.28
      },
      "latency_s": 0.2834920000168495,
      "run": "main",
      "record_index": 336
    },
    {
      "case_id": "probability-036",
      "experiment": "probability",
      "target": 0.10476190476190476,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 337,
      "started_at": "2026-10-02T08:46:26.935086+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Are both selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 12, \"unmarked\": 24}, \"event\": \"both_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.1,
      "valid": true,
      "error": null,
      "cost_usd": 3.84e-05,
      "usage": {
        "prompt_tokens": 274,
        "completion_tokens": 22,
        "total_tokens": 296,
        "cost": 3.84e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.84e-05,
          "upstream_inference_prompt_cost": 2.74e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930786-rhtNblsnQrK6frq6fzxW",
        "object": "chat.completion",
        "created": 1790930786,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 274,
          "completion_tokens": 22,
          "total_tokens": 296,
          "cost": 3.84e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.84e-05,
            "upstream_inference_prompt_cost": 2.74e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930786-rhtNblsnQrK6frq6fzxW",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.1
      },
      "latency_s": 1.0634324999991804,
      "run": "main",
      "record_index": 337
    },
    {
      "case_id": "probability-003",
      "experiment": "probability",
      "target": 0.0,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 338,
      "started_at": "2026-10-02T08:46:27.999050+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"A\": {\"marked\": 22, \"unmarked\": 53}, \"B\": {\"marked\": 0, \"unmarked\": 60}}, \"interpretation\": \"Estimate the probability of the event for this one random draw, using the supplied counts.\", \"selected_group\": \"B\", \"selection\": \"Choose one item uniformly at random from group B only.\", \"setting\": \"Fictional inventory. The following table describes the entire population, not a sample.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.91e-05,
      "usage": {
        "prompt_tokens": 291,
        "completion_tokens": 20,
        "total_tokens": 311,
        "cost": 3.91e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.91e-05,
          "upstream_inference_prompt_cost": 2.91e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930788-Q1qYjtbUzGmOIMJMGYYo",
        "object": "chat.completion",
        "created": 1790930788,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 291,
          "completion_tokens": 20,
          "total_tokens": 311,
          "cost": 3.91e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.91e-05,
            "upstream_inference_prompt_cost": 2.91e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930788-Q1qYjtbUzGmOIMJMGYYo",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.301168250007322,
      "run": "main",
      "record_index": 338
    },
    {
      "case_id": "probability-003",
      "experiment": "probability",
      "target": 0.0,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 339,
      "started_at": "2026-10-02T08:46:29.300678+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "A": {
              "marked": 22,
              "unmarked": 53
            },
            "B": {
              "marked": 0,
              "unmarked": 60
            }
          },
          "interpretation": "Estimate the probability of the event for this one random draw, using the supplied counts.",
          "selected_group": "B",
          "selection": "Choose one item uniformly at random from group B only.",
          "setting": "Fictional inventory. The following table describes the entire population, not a sample."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.01,
      "valid": true,
      "error": null,
      "cost_usd": 1.8438e-05,
      "usage": {
        "input_tokens": 439,
        "output_tokens": 20,
        "cost": 1.8438e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.01
          }
        },
        "usage": {
          "input_tokens": 439,
          "output_tokens": 20,
          "cost": 1.8438e-05
        },
        "id": "gen-dec-1790930789-ux8Tl5kzsAXxpNRtR4Iu",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930789-ux8Tl5kzsAXxpNRtR4Iu",
      "answer": {
        "type": "noul",
        "noul": 0.01
      },
      "latency_s": 0.25423070800025016,
      "run": "main",
      "record_index": 339
    },
    {
      "case_id": "boolq-dev-2665",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 340,
      "started_at": "2026-10-02T08:46:29.555422+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is there a place called cabot cove maine?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The show revolves around the day-to-day life of Jessica Fletcher, a childless, widowed, retired English teacher who becomes a successful mystery writer. Despite fame and fortune, Jessica remains a resident of Cabot Cove, a small coastal community in Maine, and maintains her links with all of her old friends, never letting her success go to her head. Exterior shots of Cabot Cove were filmed in Mendocino, California. The fictional ``Cabot Cove'' name for the series' coastal town was derived from the name of an actual bay harbor inlet in Kennebunkport, Maine, located near the town's center, on the road where motels and lobster shack dives are located.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.97,
      "valid": true,
      "error": null,
      "cost_usd": 4.57e-05,
      "usage": {
        "prompt_tokens": 347,
        "completion_tokens": 22,
        "total_tokens": 369,
        "cost": 4.57e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.57e-05,
          "upstream_inference_prompt_cost": 3.47e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930789-pFV9xZSg4gyuvDvaQRiy",
        "object": "chat.completion",
        "created": 1790930789,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.97}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 347,
          "completion_tokens": 22,
          "total_tokens": 369,
          "cost": 4.57e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.57e-05,
            "upstream_inference_prompt_cost": 3.47e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930789-pFV9xZSg4gyuvDvaQRiy",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.97
      },
      "latency_s": 0.9602852920070291,
      "run": "main",
      "record_index": 340
    },
    {
      "case_id": "boolq-dev-2665",
      "experiment": "boolq",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 341,
      "started_at": "2026-10-02T08:46:30.516163+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The show revolves around the day-to-day life of Jessica Fletcher, a childless, widowed, retired English teacher who becomes a successful mystery writer. Despite fame and fortune, Jessica remains a resident of Cabot Cove, a small coastal community in Maine, and maintains her links with all of her old friends, never letting her success go to her head. Exterior shots of Cabot Cove were filmed in Mendocino, California. The fictional ``Cabot Cove'' name for the series' coastal town was derived from the name of an actual bay harbor inlet in Kennebunkport, Maine, located near the town's center, on the road where motels and lobster shack dives are located."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is there a place called cabot cove maine?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.22,
      "valid": true,
      "error": null,
      "cost_usd": 1.9782e-05,
      "usage": {
        "input_tokens": 471,
        "output_tokens": 20,
        "cost": 1.9782e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.22
          }
        },
        "usage": {
          "input_tokens": 471,
          "output_tokens": 20,
          "cost": 1.9782e-05
        },
        "id": "gen-dec-1790930790-nRqFxOB9fjbqwhQ4bAAr",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930790-nRqFxOB9fjbqwhQ4bAAr",
      "answer": {
        "type": "noul",
        "noul": 0.22
      },
      "latency_s": 0.33818616598728113,
      "run": "main",
      "record_index": 341
    },
    {
      "case_id": "boolq-dev-1842",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 342,
      "started_at": "2026-10-02T08:46:30.854750+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is carbon a metal on the periodic table?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"Carbon (from Latin: carbo ``coal'') is a chemical element with symbol C and atomic number 6. It is nonmetallic and tetravalent--making four electrons available to form covalent chemical bonds. It belongs to group 14 of the periodic table. Three isotopes occur naturally, C and C being stable, while C is a radionuclide, decaying with a half-life of about 5,730 years. Carbon is one of the few elements known since antiquity.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.06e-05,
      "usage": {
        "prompt_tokens": 306,
        "completion_tokens": 20,
        "total_tokens": 326,
        "cost": 4.06e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.06e-05,
          "upstream_inference_prompt_cost": 3.06e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930790-YixIOjGfXd4iCSNc9G13",
        "object": "chat.completion",
        "created": 1790930790,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 306,
          "completion_tokens": 20,
          "total_tokens": 326,
          "cost": 4.06e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.06e-05,
            "upstream_inference_prompt_cost": 3.06e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930790-YixIOjGfXd4iCSNc9G13",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2651016250019893,
      "run": "main",
      "record_index": 342
    },
    {
      "case_id": "boolq-dev-1842",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 343,
      "started_at": "2026-10-02T08:46:32.120316+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "Carbon (from Latin: carbo ``coal'') is a chemical element with symbol C and atomic number 6. It is nonmetallic and tetravalent--making four electrons available to form covalent chemical bonds. It belongs to group 14 of the periodic table. Three isotopes occur naturally, C and C being stable, while C is a radionuclide, decaying with a half-life of about 5,730 years. Carbon is one of the few elements known since antiquity."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is carbon a metal on the periodic table?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.03,
      "valid": true,
      "error": null,
      "cost_usd": 1.8228e-05,
      "usage": {
        "input_tokens": 434,
        "output_tokens": 20,
        "cost": 1.8228e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.03
          }
        },
        "usage": {
          "input_tokens": 434,
          "output_tokens": 20,
          "cost": 1.8228e-05
        },
        "id": "gen-dec-1790930792-BmmkwbIqkEkfHZfbQzUz",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930792-BmmkwbIqkEkfHZfbQzUz",
      "answer": {
        "type": "noul",
        "noul": 0.03
      },
      "latency_s": 0.3519619590078946,
      "run": "main",
      "record_index": 343
    },
    {
      "case_id": "probability-008",
      "experiment": "probability",
      "target": 0.9692307692307692,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 344,
      "started_at": "2026-10-02T08:46:32.472663+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is at least one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 21, \"unmarked\": 5}, \"event\": \"at_least_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.9677,
      "valid": true,
      "error": null,
      "cost_usd": 3.97e-05,
      "usage": {
        "prompt_tokens": 282,
        "completion_tokens": 23,
        "total_tokens": 305,
        "cost": 3.97e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.97e-05,
          "upstream_inference_prompt_cost": 2.82e-05,
          "upstream_inference_completions_cost": 1.15e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930792-ROxoxSIfT9y7YQg0AI6l",
        "object": "chat.completion",
        "created": 1790930792,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.9677}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 282,
          "completion_tokens": 23,
          "total_tokens": 305,
          "cost": 3.97e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.97e-05,
            "upstream_inference_prompt_cost": 2.82e-05,
            "upstream_inference_completions_cost": 1.15e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930792-ROxoxSIfT9y7YQg0AI6l",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.9677
      },
      "latency_s": 1.4479606249951757,
      "run": "main",
      "record_index": 344
    },
    {
      "case_id": "probability-008",
      "experiment": "probability",
      "target": 0.9692307692307692,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 345,
      "started_at": "2026-10-02T08:46:33.921132+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 21,
            "unmarked": 5
          },
          "event": "at_least_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is at least one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.96,
      "valid": true,
      "error": null,
      "cost_usd": 1.764e-05,
      "usage": {
        "input_tokens": 420,
        "output_tokens": 20,
        "cost": 1.764e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.96
          }
        },
        "usage": {
          "input_tokens": 420,
          "output_tokens": 20,
          "cost": 1.764e-05
        },
        "id": "gen-dec-1790930793-5QCEqZv9hY1CHryLjfnH",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930793-5QCEqZv9hY1CHryLjfnH",
      "answer": {
        "type": "noul",
        "noul": 0.96
      },
      "latency_s": 0.2202720000059344,
      "run": "main",
      "record_index": 345
    },
    {
      "case_id": "probability-047",
      "experiment": "probability",
      "target": 0.7708333333333334,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 346,
      "started_at": "2026-10-02T08:46:34.142184+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is the selected item marked?\", \"type\": \"noul\"}}, \"state\": {\"box_selection_tickets\": {\"A\": 3, \"B\": 6}, \"boxes\": {\"A\": {\"marked\": 5, \"unmarked\": 11}, \"B\": {\"marked\": 27, \"unmarked\": 0}}, \"interpretation\": \"Estimate the probability of the event for this random item, accounting for both stages of selection.\", \"selection\": \"First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.\", \"setting\": \"Two fictional boxes contain complete populations. Their contents are listed below.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.8,
      "valid": true,
      "error": null,
      "cost_usd": 4.39e-05,
      "usage": {
        "prompt_tokens": 329,
        "completion_tokens": 22,
        "total_tokens": 351,
        "cost": 4.39e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.39e-05,
          "upstream_inference_prompt_cost": 3.29e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930794-Uvy1Z4jyVequDdDHOnA1",
        "object": "chat.completion",
        "created": 1790930794,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.8}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 329,
          "completion_tokens": 22,
          "total_tokens": 351,
          "cost": 4.39e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.39e-05,
            "upstream_inference_prompt_cost": 3.29e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930794-Uvy1Z4jyVequDdDHOnA1",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.8
      },
      "latency_s": 1.0749241250159685,
      "run": "main",
      "record_index": 346
    },
    {
      "case_id": "probability-047",
      "experiment": "probability",
      "target": 0.7708333333333334,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 347,
      "started_at": "2026-10-02T08:46:35.217474+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "box_selection_tickets": {
            "A": 3,
            "B": 6
          },
          "boxes": {
            "A": {
              "marked": 5,
              "unmarked": 11
            },
            "B": {
              "marked": 27,
              "unmarked": 0
            }
          },
          "interpretation": "Estimate the probability of the event for this random item, accounting for both stages of selection.",
          "selection": "First choose one ticket uniformly at random from all tickets. The ticket names box A or box B. Then choose one item uniformly at random from that box. There is no further information about the chosen box.",
          "setting": "Two fictional boxes contain complete populations. Their contents are listed below."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is the selected item marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.73,
      "valid": true,
      "error": null,
      "cost_usd": 1.9992e-05,
      "usage": {
        "input_tokens": 476,
        "output_tokens": 20,
        "cost": 1.9992e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.73
          }
        },
        "usage": {
          "input_tokens": 476,
          "output_tokens": 20,
          "cost": 1.9992e-05
        },
        "id": "gen-dec-1790930795-tfjhu6eZSdmNxecEhHM8",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930795-tfjhu6eZSdmNxecEhHM8",
      "answer": {
        "type": "noul",
        "noul": 0.73
      },
      "latency_s": 0.22215562500059605,
      "run": "main",
      "record_index": 347
    },
    {
      "case_id": "policy-024",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 348,
      "started_at": "2026-10-02T08:46:35.439960+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.",
          "record": {
            "customer_member": false,
            "days_since_purchase": 120,
            "defective": true,
            "final_sale": false,
            "price_credits": 73,
            "recalled": false,
            "receipt": true,
            "unused": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this item eligible for a refund under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 2.0412e-05,
      "usage": {
        "input_tokens": 486,
        "output_tokens": 20,
        "cost": 2.0412e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 486,
          "output_tokens": 20,
          "cost": 2.0412e-05
        },
        "id": "gen-dec-1790930795-wTgtVV4oxMqAyLb7U36W",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930795-wTgtVV4oxMqAyLb7U36W",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.21910479202051647,
      "run": "main",
      "record_index": 348
    },
    {
      "case_id": "policy-024",
      "experiment": "policy",
      "target": 0.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 349,
      "started_at": "2026-10-02T08:46:35.659422+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this item eligible for a refund under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional shop policy. A recalled item is eligible for a refund regardless of all other facts. For an item that is not recalled, a receipt is required and final-sale items are never eligible. With those requirements met, an item is eligible if it is defective and purchased at most 90 days ago, or if it is unused and purchased at most 30 days ago. Boundaries are inclusive. No other exceptions apply. The item price and customer membership do not affect eligibility.\", \"record\": {\"customer_member\": false, \"days_since_purchase\": 120, \"defective\": true, \"final_sale\": false, \"price_credits\": 73, \"recalled\": false, \"receipt\": true, \"unused\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.46e-05,
      "usage": {
        "prompt_tokens": 346,
        "completion_tokens": 20,
        "total_tokens": 366,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 3.46e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930795-Vk0K318Q1VBFQULpQNLE",
        "object": "chat.completion",
        "created": 1790930795,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 346,
          "completion_tokens": 20,
          "total_tokens": 366,
          "cost": 4.46e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.46e-05,
            "upstream_inference_prompt_cost": 3.46e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930795-Vk0K318Q1VBFQULpQNLE",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0492995829845313,
      "run": "main",
      "record_index": 349
    },
    {
      "case_id": "boolq-dev-0468",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 350,
      "started_at": "2026-10-02T08:46:36.709130+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The Great Pacific garbage patch, also described as the Pacific trash vortex, is a gyre of marine debris particles in the central North Pacific Ocean discovered between 1985 and 1988. It is located roughly between 135°W to 155°W and 35°N to 42°N. The collection of plastic, floating trash halfway between Hawaii and California extends over an indeterminate area of widely varying range depending on the degree of plastic concentration used to define the affected area."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is there a floating mass of garbage in the ocean?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 1.848e-05,
      "usage": {
        "input_tokens": 440,
        "output_tokens": 20,
        "cost": 1.848e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.99
          }
        },
        "usage": {
          "input_tokens": 440,
          "output_tokens": 20,
          "cost": 1.848e-05
        },
        "id": "gen-dec-1790930796-M96hIQZos3KBWulLjjEU",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930796-M96hIQZos3KBWulLjjEU",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 0.2252970409754198,
      "run": "main",
      "record_index": 350
    },
    {
      "case_id": "boolq-dev-0468",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 351,
      "started_at": "2026-10-02T08:46:36.934816+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is there a floating mass of garbage in the ocean?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The Great Pacific garbage patch, also described as the Pacific trash vortex, is a gyre of marine debris particles in the central North Pacific Ocean discovered between 1985 and 1988. It is located roughly between 135°W to 155°W and 35°N to 42°N. The collection of plastic, floating trash halfway between Hawaii and California extends over an indeterminate area of widely varying range depending on the degree of plastic concentration used to define the affected area.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.99,
      "valid": true,
      "error": null,
      "cost_usd": 4.16e-05,
      "usage": {
        "prompt_tokens": 306,
        "completion_tokens": 22,
        "total_tokens": 328,
        "cost": 4.16e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.16e-05,
          "upstream_inference_prompt_cost": 3.06e-05,
          "upstream_inference_completions_cost": 1.1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930796-pEeOalfUNHwlAii8ISoB",
        "object": "chat.completion",
        "created": 1790930796,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.99}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 306,
          "completion_tokens": 22,
          "total_tokens": 328,
          "cost": 4.16e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.16e-05,
            "upstream_inference_prompt_cost": 3.06e-05,
            "upstream_inference_completions_cost": 1.1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930796-pEeOalfUNHwlAii8ISoB",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.99
      },
      "latency_s": 1.1362126670137513,
      "run": "main",
      "record_index": 351
    },
    {
      "case_id": "boolq-dev-0587",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 352,
      "started_at": "2026-10-02T08:46:38.071475+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "At the 2010 Consumer Electronics Show, Boost Mobile announced it would begin to offer a new unlimited plan using Sprint's CDMA network, costing $50 a month. For $10 more, Boost also offered an unlimited plan for the BlackBerry Curve 8830. Sprint would also acquire fellow prepaid wireless provider Virgin Mobile USA in 2010--both Boost and Virgin Mobile would be re-organized into a new group within Sprint, encompassing the two brands and other no-contract phone services offered by the company."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "is boost mobile and virgin mobile the same?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.15,
      "valid": true,
      "error": null,
      "cost_usd": 1.8732e-05,
      "usage": {
        "input_tokens": 446,
        "output_tokens": 20,
        "cost": 1.8732e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.15
          }
        },
        "usage": {
          "input_tokens": 446,
          "output_tokens": 20,
          "cost": 1.8732e-05
        },
        "id": "gen-dec-1790930798-xJRnreytUp3gIftXNvG1",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930798-xJRnreytUp3gIftXNvG1",
      "answer": {
        "type": "noul",
        "noul": 0.15
      },
      "latency_s": 0.23421241700998507,
      "run": "main",
      "record_index": 352
    },
    {
      "case_id": "boolq-dev-0587",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 353,
      "started_at": "2026-10-02T08:46:38.306018+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"is boost mobile and virgin mobile the same?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"At the 2010 Consumer Electronics Show, Boost Mobile announced it would begin to offer a new unlimited plan using Sprint's CDMA network, costing $50 a month. For $10 more, Boost also offered an unlimited plan for the BlackBerry Curve 8830. Sprint would also acquire fellow prepaid wireless provider Virgin Mobile USA in 2010--both Boost and Virgin Mobile would be re-organized into a new group within Sprint, encompassing the two brands and other no-contract phone services offered by the company.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.07e-05,
      "usage": {
        "prompt_tokens": 307,
        "completion_tokens": 20,
        "total_tokens": 327,
        "cost": 4.07e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.07e-05,
          "upstream_inference_prompt_cost": 3.07e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930798-fUB13muhCIWGnoJ6ikaP",
        "object": "chat.completion",
        "created": 1790930798,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 307,
          "completion_tokens": 20,
          "total_tokens": 327,
          "cost": 4.07e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.07e-05,
            "upstream_inference_prompt_cost": 3.07e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930798-fUB13muhCIWGnoJ6ikaP",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.318288999987999,
      "run": "main",
      "record_index": 353
    },
    {
      "case_id": "boolq-dev-2630",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 354,
      "started_at": "2026-10-02T08:46:39.624868+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the question is no, using the supplied passage as evidence.\", \"true\": \"The answer to the question is yes, using the supplied passage as evidence.\"}, \"instructions\": \"are chickpeas and garbonzo beans the same thing?\", \"type\": \"noul\"}}, \"state\": {\"passage\": \"The chickpea or chick pea (Cicer arietinum) is a legume of the family Fabaceae, subfamily Faboideae. Its different types are variously known as gram or Bengal gram, garbanzo or garbanzo bean, and Egyptian pea. Chickpea seeds are high in protein. It is one of the earliest cultivated legumes and 7500-year-old remains have been found in the Middle East.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3.95e-05,
      "usage": {
        "prompt_tokens": 295,
        "completion_tokens": 20,
        "total_tokens": 315,
        "cost": 3.95e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.95e-05,
          "upstream_inference_prompt_cost": 2.95e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930799-vqjiMtPXKzp44rAkLLc4",
        "object": "chat.completion",
        "created": 1790930799,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 295,
          "completion_tokens": 20,
          "total_tokens": 315,
          "cost": 3.95e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.95e-05,
            "upstream_inference_prompt_cost": 2.95e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930799-vqjiMtPXKzp44rAkLLc4",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9916192500095349,
      "run": "main",
      "record_index": 354
    },
    {
      "case_id": "boolq-dev-2630",
      "experiment": "boolq",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 355,
      "started_at": "2026-10-02T08:46:40.616872+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "passage": "The chickpea or chick pea (Cicer arietinum) is a legume of the family Fabaceae, subfamily Faboideae. Its different types are variously known as gram or Bengal gram, garbanzo or garbanzo bean, and Egyptian pea. Chickpea seeds are high in protein. It is one of the earliest cultivated legumes and 7500-year-old remains have been found in the Middle East."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "are chickpeas and garbonzo beans the same thing?",
            "criteria": {
              "false": "The answer to the question is no, using the supplied passage as evidence.",
              "true": "The answer to the question is yes, using the supplied passage as evidence."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 1.7934e-05,
      "usage": {
        "input_tokens": 427,
        "output_tokens": 20,
        "cost": 1.7934e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 427,
          "output_tokens": 20,
          "cost": 1.7934e-05
        },
        "id": "gen-dec-1790930800-r0gOIAt65eKxH6WKhPPD",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930800-r0gOIAt65eKxH6WKhPPD",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.2403121670067776,
      "run": "main",
      "record_index": 355
    },
    {
      "case_id": "probability-019",
      "experiment": "probability",
      "target": 0.9333333333333333,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 356,
      "started_at": "2026-10-02T08:46:40.857520+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 7,
            "unmarked": 3
          },
          "event": "at_least_one_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is at least one of the two selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.91,
      "valid": true,
      "error": null,
      "cost_usd": 1.7598e-05,
      "usage": {
        "input_tokens": 419,
        "output_tokens": 20,
        "cost": 1.7598e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.91
          }
        },
        "usage": {
          "input_tokens": 419,
          "output_tokens": 20,
          "cost": 1.7598e-05
        },
        "id": "gen-dec-1790930800-hM1YwQ4FYWy7JARLvXRl",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930800-hM1YwQ4FYWy7JARLvXRl",
      "answer": {
        "type": "noul",
        "noul": 0.91
      },
      "latency_s": 0.26699479200760834,
      "run": "main",
      "record_index": 356
    },
    {
      "case_id": "probability-019",
      "experiment": "probability",
      "target": 0.9333333333333333,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 357,
      "started_at": "2026-10-02T08:46:41.124861+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Is at least one of the two selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 7, \"unmarked\": 3}, \"event\": \"at_least_one_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.9333333333,
      "valid": true,
      "error": null,
      "cost_usd": 4.07e-05,
      "usage": {
        "prompt_tokens": 282,
        "completion_tokens": 25,
        "total_tokens": 307,
        "cost": 4.07e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.07e-05,
          "upstream_inference_prompt_cost": 2.82e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930801-GYHaMfAz2OnlymwjypeQ",
        "object": "chat.completion",
        "created": 1790930801,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.9333333333}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 282,
          "completion_tokens": 25,
          "total_tokens": 307,
          "cost": 4.07e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.07e-05,
            "upstream_inference_prompt_cost": 2.82e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930801-GYHaMfAz2OnlymwjypeQ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.9333333333
      },
      "latency_s": 2.4705519579874817,
      "run": "main",
      "record_index": 357
    },
    {
      "case_id": "policy-037",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 358,
      "started_at": "2026-10-02T08:46:43.595955+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.",
          "record": {
            "account_active": true,
            "badge_color": "green",
            "clearance": 3,
            "emergency_permit": true,
            "experience_years": 0,
            "role": "contractor",
            "sponsor": true,
            "suspended": false,
            "training_complete": false
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is this person allowed laboratory access under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.98,
      "valid": true,
      "error": null,
      "cost_usd": 2.1084e-05,
      "usage": {
        "input_tokens": 502,
        "output_tokens": 20,
        "cost": 2.1084e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.98
          }
        },
        "usage": {
          "input_tokens": 502,
          "output_tokens": 20,
          "cost": 2.1084e-05
        },
        "id": "gen-dec-1790930803-pp2tLygu7nBIMyfTTq58",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930803-pp2tLygu7nBIMyfTTq58",
      "answer": {
        "type": "noul",
        "noul": 0.98
      },
      "latency_s": 0.2702599170152098,
      "run": "main",
      "record_index": 358
    },
    {
      "case_id": "policy-037",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 359,
      "started_at": "2026-10-02T08:46:43.884212+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Is this person allowed laboratory access under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional laboratory access policy. The account must be active and must not be suspended. These two requirements always apply. An emergency permit then grants access regardless of role, training, clearance, or sponsor. Without an emergency permit, training must be complete, and the person must either be staff with clearance at least 2 or be a contractor with clearance at least 3 and a sponsor. Visitors never qualify without an emergency permit. Boundaries are inclusive. Years of experience and the badge color do not affect access.\", \"record\": {\"account_active\": true, \"badge_color\": \"green\", \"clearance\": 3, \"emergency_permit\": true, \"experience_years\": 0, \"role\": \"contractor\", \"sponsor\": true, \"suspended\": false, \"training_complete\": false}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.62e-05,
      "usage": {
        "prompt_tokens": 362,
        "completion_tokens": 20,
        "total_tokens": 382,
        "cost": 4.62e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.62e-05,
          "upstream_inference_prompt_cost": 3.62e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930803-3aCdN50q2KwiK9FW8u8x",
        "object": "chat.completion",
        "created": 1790930803,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 362,
          "completion_tokens": 20,
          "total_tokens": 382,
          "cost": 4.62e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.62e-05,
            "upstream_inference_prompt_cost": 3.62e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930803-3aCdN50q2KwiK9FW8u8x",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0011180840083398,
      "run": "main",
      "record_index": 359
    },
    {
      "case_id": "probability-054",
      "experiment": "probability",
      "target": 0.023809523809523808,
      "target_kind": "probability",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 360,
      "started_at": "2026-10-02T08:46:44.885721+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "counts": {
            "marked": 6,
            "unmarked": 30
          },
          "event": "both_marked",
          "interpretation": "Estimate the probability of the event for these two random draws, using the supplied counts.",
          "selection": "Select two different items uniformly at random without replacement. Every unordered pair is equally likely.",
          "setting": "A fictional box contains the entire population described by these counts."
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Are both selected items marked?",
            "criteria": {
              "false": "The stated event does not occur in that random selection.",
              "true": "The stated event occurs in the random selection described in the state."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.15,
      "valid": true,
      "error": null,
      "cost_usd": 1.7304e-05,
      "usage": {
        "input_tokens": 412,
        "output_tokens": 20,
        "cost": 1.7304e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.15
          }
        },
        "usage": {
          "input_tokens": 412,
          "output_tokens": 20,
          "cost": 1.7304e-05
        },
        "id": "gen-dec-1790930804-ZZxJClThzWu7mYEW3h8U",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930804-ZZxJClThzWu7mYEW3h8U",
      "answer": {
        "type": "noul",
        "noul": 0.15
      },
      "latency_s": 0.24594249998335727,
      "run": "main",
      "record_index": 360
    },
    {
      "case_id": "probability-054",
      "experiment": "probability",
      "target": 0.023809523809523808,
      "target_kind": "probability",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 361,
      "started_at": "2026-10-02T08:46:45.132122+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The stated event does not occur in that random selection.\", \"true\": \"The stated event occurs in the random selection described in the state.\"}, \"instructions\": \"Are both selected items marked?\", \"type\": \"noul\"}}, \"state\": {\"counts\": {\"marked\": 6, \"unmarked\": 30}, \"event\": \"both_marked\", \"interpretation\": \"Estimate the probability of the event for these two random draws, using the supplied counts.\", \"selection\": \"Select two different items uniformly at random without replacement. Every unordered pair is equally likely.\", \"setting\": \"A fictional box contains the entire population described by these counts.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0238095238,
      "valid": true,
      "error": null,
      "cost_usd": 3.99e-05,
      "usage": {
        "prompt_tokens": 274,
        "completion_tokens": 25,
        "total_tokens": 299,
        "cost": 3.99e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.99e-05,
          "upstream_inference_prompt_cost": 2.74e-05,
          "upstream_inference_completions_cost": 1.25e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930805-DsKrLLxwLukKZZvOFv8G",
        "object": "chat.completion",
        "created": 1790930805,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0.0238095238}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 274,
          "completion_tokens": 25,
          "total_tokens": 299,
          "cost": 3.99e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3.99e-05,
            "upstream_inference_prompt_cost": 2.74e-05,
            "upstream_inference_completions_cost": 1.25e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930805-DsKrLLxwLukKZZvOFv8G",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0.0238095238
      },
      "latency_s": 1.0233090420078952,
      "run": "main",
      "record_index": 361
    },
    {
      "case_id": "policy-042",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 362,
      "started_at": "2026-10-02T08:46:46.155841+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The answer to the stated question is no.\", \"true\": \"The answer to the stated question is yes.\"}, \"instructions\": \"Does this order qualify for free delivery under the supplied policy?\", \"type\": \"noul\"}}, \"state\": {\"policy\": \"This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.\", \"record\": {\"account_age_days\": 1767, \"active_promotion\": false, \"gift_wrapping\": false, \"hazardous\": false, \"mainland\": true, \"member\": true, \"subtotal_credits\": 99, \"weight_kg\": 20}}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 4.65e-05,
      "usage": {
        "prompt_tokens": 365,
        "completion_tokens": 20,
        "total_tokens": 385,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 3.65e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930806-tXFcFGUGZ5z2BDOMyBFJ",
        "object": "chat.completion",
        "created": 1790930806,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 365,
          "completion_tokens": 20,
          "total_tokens": 385,
          "cost": 4.65e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 4.65e-05,
            "upstream_inference_prompt_cost": 3.65e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930806-tXFcFGUGZ5z2BDOMyBFJ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1742852089810185,
      "run": "main",
      "record_index": 362
    },
    {
      "case_id": "policy-042",
      "experiment": "policy",
      "target": 1.0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 363,
      "started_at": "2026-10-02T08:46:47.330640+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "policy": "This is a fictional delivery policy. Free delivery is available only for mainland destinations, non-hazardous orders, and total weight at most 20 kilograms. These requirements cannot be waived. An order meeting them qualifies if its subtotal is at least 100 credits, or if the customer is a member and the subtotal is at least 50 credits, or if the order has an active free-delivery promotion. An active promotion waives only the subtotal and membership requirements. Boundaries are inclusive. Gift wrapping and customer account age do not affect the decision.",
          "record": {
            "account_age_days": 1767,
            "active_promotion": false,
            "gift_wrapping": false,
            "hazardous": false,
            "mainland": true,
            "member": true,
            "subtotal_credits": 99,
            "weight_kg": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Does this order qualify for free delivery under the supplied policy?",
            "criteria": {
              "false": "The answer to the stated question is no.",
              "true": "The answer to the stated question is yes."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.6,
      "valid": true,
      "error": null,
      "cost_usd": 2.1294e-05,
      "usage": {
        "input_tokens": 507,
        "output_tokens": 20,
        "cost": 2.1294e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.6
          }
        },
        "usage": {
          "input_tokens": 507,
          "output_tokens": 20,
          "cost": 2.1294e-05
        },
        "id": "gen-dec-1790930807-6wvxEYnpKlFH6Iqn60nP",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930807-6wvxEYnpKlFH6Iqn60nP",
      "answer": {
        "type": "noul",
        "noul": 0.6
      },
      "latency_s": 0.24411987501662225,
      "run": "main",
      "record_index": 363
    },
    {
      "case_id": "cache-prime-unique",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "cache_prime",
      "sequence": 0,
      "started_at": "2026-10-02T08:49:58.827271+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 77, \"calibration_current\": true, \"category\": \"portable\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 3, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 6, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.000362525,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.000362525,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 2533,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000362525,
          "upstream_inference_prompt_cost": 0.000352525,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930999-Yd0AT0glOnbvPxuPJWPx",
        "object": "chat.completion",
        "created": 1790930999,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.000362525,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 2533,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.000362525,
            "upstream_inference_prompt_cost": 0.000352525,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930999-Yd0AT0glOnbvPxuPJWPx",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.2044517089962028,
      "cached_tokens": 0,
      "cache_write_tokens": 2533,
      "run": "cache",
      "record_index": 364
    },
    {
      "case_id": "cache-after-hours-permitted",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 1,
      "started_at": "2026-10-02T08:50:00.032558+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": true, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"rotating\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 2, \"duty_manager_signed\": true, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 2, \"pickup_staffed\": false, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931000-kIzQuGSKiqJ6YCWazzST",
        "object": "chat.completion",
        "created": 1790931000,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931000-kIzQuGSKiqJ6YCWazzST",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9644522090093233,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 365
    },
    {
      "case_id": "cache-after-hours-permitted",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 2,
      "started_at": "2026-10-02T08:50:00.997834+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": true, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"rotating\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 2, \"duty_manager_signed\": true, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 2, \"pickup_staffed\": false, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931001-1rKUthaev334YQ9zd31R",
        "object": "chat.completion",
        "created": 1790931001,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931001-1rKUthaev334YQ9zd31R",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9355357500026003,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 366
    },
    {
      "case_id": "cache-after-hours-permitted",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 3,
      "started_at": "2026-10-02T08:50:01.934010+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "rotating",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": true,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": false,
            "after_hours_permit": true,
            "notice_hours": 2,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 2,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.84,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132678,
      "usage": {
        "input_tokens": 3159,
        "output_tokens": 20,
        "cost": 0.000132678
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.84
          }
        },
        "usage": {
          "input_tokens": 3159,
          "output_tokens": 20,
          "cost": 0.000132678
        },
        "id": "gen-dec-1790931001-Iig3YfS8sBlyBVVFZili",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931001-Iig3YfS8sBlyBVVFZili",
      "answer": {
        "type": "noul",
        "noul": 0.84
      },
      "latency_s": 0.25161004200344905,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 367
    },
    {
      "case_id": "cache-sensor-outdoor-boundary",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 4,
      "started_at": "2026-10-02T08:50:02.186480+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 20, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": false, \"duration_hours\": 72, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": -20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 3000, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931002-1BpOnv401z3Ai8N0WDAq",
        "object": "chat.completion",
        "created": 1790931002,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931002-1BpOnv401z3Ai8N0WDAq",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0961741670034826,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 368
    },
    {
      "case_id": "cache-sensor-outdoor-boundary",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 5,
      "started_at": "2026-10-02T08:50:03.283315+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "sensor",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": false,
            "weather_clear": true,
            "transport_distance_m": 3000,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 72,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": -20,
            "battery_percent": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.83,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132762,
      "usage": {
        "input_tokens": 3161,
        "output_tokens": 20,
        "cost": 0.000132762
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.83
          }
        },
        "usage": {
          "input_tokens": 3161,
          "output_tokens": 20,
          "cost": 0.000132762
        },
        "id": "gen-dec-1790931003-owuYQkSkLYhQzLp7k2QF",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931003-owuYQkSkLYhQzLp7k2QF",
      "answer": {
        "type": "noul",
        "noul": 0.83
      },
      "latency_s": 0.38876874998095445,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 369
    },
    {
      "case_id": "cache-sensor-outdoor-boundary",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 6,
      "started_at": "2026-10-02T08:50:03.672839+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 20, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": false, \"duration_hours\": 72, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": -20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 3000, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931003-vXOLlHVi2SKReDorsKsE",
        "object": "chat.completion",
        "created": 1790931003,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931003-vXOLlHVi2SKReDorsKsE",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9754823750117794,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 370
    },
    {
      "case_id": "cache-portable-weather-irrelevant",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 7,
      "started_at": "2026-10-02T08:50:04.649066+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 30, \"calibration_current\": false, \"category\": \"portable\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": false, \"duration_hours\": 24, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": false, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": false}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002991,
      "usage": {
        "prompt_tokens": 2891,
        "completion_tokens": 20,
        "total_tokens": 2911,
        "cost": 0.0002991,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002991,
          "upstream_inference_prompt_cost": 0.0002891,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931004-b9p1CaJcEtVkC6MQaZfK",
        "object": "chat.completion",
        "created": 1790931004,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2891,
          "completion_tokens": 20,
          "total_tokens": 2911,
          "cost": 0.0002991,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002991,
            "upstream_inference_prompt_cost": 0.0002891,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931004-b9p1CaJcEtVkC6MQaZfK",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.05280216599931,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 371
    },
    {
      "case_id": "cache-portable-weather-irrelevant",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 8,
      "started_at": "2026-10-02T08:50:05.702675+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 30, \"calibration_current\": false, \"category\": \"portable\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": false, \"duration_hours\": 24, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": false, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": false}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.123e-05,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 7.123e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.123e-05,
          "upstream_inference_prompt_cost": 6.123e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931005-XEq5rLXJg5BLHJn4M2kg",
        "object": "chat.completion",
        "created": 1790931005,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 7.123e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.123e-05,
            "upstream_inference_prompt_cost": 6.123e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931005-XEq5rLXJg5BLHJn4M2kg",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.009998291003285,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 372
    },
    {
      "case_id": "cache-portable-weather-irrelevant",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 9,
      "started_at": "2026-10-02T08:50:06.713369+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "portable",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": false,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": false,
            "weather_clear": false,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 24,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": false,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 30
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.94,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132678,
      "usage": {
        "input_tokens": 3159,
        "output_tokens": 20,
        "cost": 0.000132678
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.94
          }
        },
        "usage": {
          "input_tokens": 3159,
          "output_tokens": 20,
          "cost": 0.000132678
        },
        "id": "gen-dec-1790931006-kRf5xGtFLH9bvxPSfPId",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931006-kRf5xGtFLH9bvxPSfPId",
      "answer": {
        "type": "noul",
        "noul": 0.94
      },
      "latency_s": 0.29784762498456985,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 373
    },
    {
      "case_id": "cache-trainee-remote",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 10,
      "started_at": "2026-10-02T08:50:07.011810+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "portable",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "trainee",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": true,
            "supervisor_present": false,
            "supervisor_category_endorsed": true,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.15,
      "valid": true,
      "error": null,
      "cost_usd": 0.00013272,
      "usage": {
        "input_tokens": 3160,
        "output_tokens": 20,
        "cost": 0.00013272
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.15
          }
        },
        "usage": {
          "input_tokens": 3160,
          "output_tokens": 20,
          "cost": 0.00013272
        },
        "id": "gen-dec-1790931007-1J77WvdfnqKeNCTA2cWl",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931007-1J77WvdfnqKeNCTA2cWl",
      "answer": {
        "type": "noul",
        "noul": 0.15
      },
      "latency_s": 0.2611688330071047,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 374
    },
    {
      "case_id": "cache-trainee-remote",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 11,
      "started_at": "2026-10-02T08:50:07.273723+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"portable\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"trainee\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": true, \"supervisor_category_endorsed\": true, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931007-c5lskZAOiwLWVNpUSAaD",
        "object": "chat.completion",
        "created": 1790931007,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931007-c5lskZAOiwLWVNpUSAaD",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9643918749934528,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 375
    },
    {
      "case_id": "cache-trainee-remote",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 12,
      "started_at": "2026-10-02T08:50:08.238890+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"portable\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"trainee\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": true, \"supervisor_category_endorsed\": true, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931008-IeEH8idzo1i08f3tcrpg",
        "object": "chat.completion",
        "created": 1790931008,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931008-IeEH8idzo1i08f3tcrpg",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.1919745420163963,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 376
    },
    {
      "case_id": "cache-optical-no-transport",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 13,
      "started_at": "2026-10-02T08:50:09.431622+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"optical\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": false, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 10, \"fire_watch\": false, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": false, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 0, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931009-0w1B8IGOVaLCdisVy8nz",
        "object": "chat.completion",
        "created": 1790931009,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931009-0w1B8IGOVaLCdisVy8nz",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9566210829943884,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 377
    },
    {
      "case_id": "cache-optical-no-transport",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 14,
      "started_at": "2026-10-02T08:50:10.388918+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "optical",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 0,
            "protective_case": false,
            "clean_route": false,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": false,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 10,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.91,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132636,
      "usage": {
        "input_tokens": 3158,
        "output_tokens": 20,
        "cost": 0.000132636
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.91
          }
        },
        "usage": {
          "input_tokens": 3158,
          "output_tokens": 20,
          "cost": 0.000132636
        },
        "id": "gen-dec-1790931010-WVhuScxk3o1Sc1PObJma",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931010-WVhuScxk3o1Sc1PObJma",
      "answer": {
        "type": "noul",
        "noul": 0.91
      },
      "latency_s": 0.25992766598938033,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 378
    },
    {
      "case_id": "cache-optical-no-transport",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 15,
      "started_at": "2026-10-02T08:50:10.649572+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"optical\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": false, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 10, \"fire_watch\": false, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": false, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 0, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931010-b8bGNoFhpJndwvogwXAM",
        "object": "chat.completion",
        "created": 1790931010,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931010-b8bGNoFhpJndwvogwXAM",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.8939200000022538,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 379
    },
    {
      "case_id": "cache-induction-expired",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 16,
      "started_at": "2026-10-02T08:50:11.544148+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "optical",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 366,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": true,
            "duty_manager_signed": true,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.05,
      "valid": true,
      "error": null,
      "cost_usd": 0.00013272,
      "usage": {
        "input_tokens": 3160,
        "output_tokens": 20,
        "cost": 0.00013272
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.05
          }
        },
        "usage": {
          "input_tokens": 3160,
          "output_tokens": 20,
          "cost": 0.00013272
        },
        "id": "gen-dec-1790931011-evgbR1kqJNEd47GDDaW9",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931011-evgbR1kqJNEd47GDDaW9",
      "answer": {
        "type": "noul",
        "noul": 0.05
      },
      "latency_s": 0.24939220800297335,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 380
    },
    {
      "case_id": "cache-induction-expired",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 17,
      "started_at": "2026-10-02T08:50:11.794321+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"optical\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 366, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931011-riSUgL3Xg7bigz5HOUF7",
        "object": "chat.completion",
        "created": 1790931011,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931011-riSUgL3Xg7bigz5HOUF7",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.002654000010807,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 381
    },
    {
      "case_id": "cache-induction-expired",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 18,
      "started_at": "2026-10-02T08:50:12.797697+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"optical\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 366, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931012-OMivjO9Ojm591NW72Hk8",
        "object": "chat.completion",
        "created": 1790931012,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931012-OMivjO9Ojm591NW72Hk8",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.8828063330147415,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 382
    },
    {
      "case_id": "cache-member-supervised-expired",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 19,
      "started_at": "2026-10-02T08:50:13.681316+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 5, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 181, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": true, \"supervisor_category_endorsed\": true, \"supervisor_present\": true, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.123e-05,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 7.123e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.123e-05,
          "upstream_inference_prompt_cost": 6.123e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931013-vKPPywaaX3lt7SdJqlxu",
        "object": "chat.completion",
        "created": 1790931013,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 7.123e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.123e-05,
            "upstream_inference_prompt_cost": 6.123e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931013-vKPPywaaX3lt7SdJqlxu",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.8233834999846295,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 383
    },
    {
      "case_id": "cache-member-supervised-expired",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 20,
      "started_at": "2026-10-02T08:50:14.505341+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "sensor",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 181,
            "supervisor_active": true,
            "supervisor_present": true,
            "supervisor_category_endorsed": true,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 5,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.69,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132678,
      "usage": {
        "input_tokens": 3159,
        "output_tokens": 20,
        "cost": 0.000132678
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.69
          }
        },
        "usage": {
          "input_tokens": 3159,
          "output_tokens": 20,
          "cost": 0.000132678
        },
        "id": "gen-dec-1790931014-2zbVF1clfGs90VC7juwL",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931014-2zbVF1clfGs90VC7juwL",
      "answer": {
        "type": "noul",
        "noul": 0.69
      },
      "latency_s": 0.2780308750225231,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 384
    },
    {
      "case_id": "cache-member-supervised-expired",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 21,
      "started_at": "2026-10-02T08:50:14.784110+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 5, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 181, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": true, \"supervisor_category_endorsed\": true, \"supervisor_present\": true, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002991,
      "usage": {
        "prompt_tokens": 2891,
        "completion_tokens": 20,
        "total_tokens": 2911,
        "cost": 0.0002991,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002991,
          "upstream_inference_prompt_cost": 0.0002891,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931014-D03RT65qYFngZZapQd0g",
        "object": "chat.completion",
        "created": 1790931014,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2891,
          "completion_tokens": 20,
          "total_tokens": 2911,
          "cost": 0.0002991,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002991,
            "upstream_inference_prompt_cost": 0.0002891,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931014-D03RT65qYFngZZapQd0g",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.179873332992429,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 385
    },
    {
      "case_id": "cache-maintenance-last-day",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 22,
      "started_at": "2026-10-02T08:50:15.964753+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 2, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"thermal\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 35, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 1, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.123e-05,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 7.123e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.123e-05,
          "upstream_inference_prompt_cost": 6.123e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931015-KDyCmbF2H1Zl5JqeCnXP",
        "object": "chat.completion",
        "created": 1790931015,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 7.123e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.123e-05,
            "upstream_inference_prompt_cost": 6.123e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931015-KDyCmbF2H1Zl5JqeCnXP",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.251140333013609,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 386
    },
    {
      "case_id": "cache-maintenance-last-day",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 23,
      "started_at": "2026-10-02T08:50:17.216625+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "thermal",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 1,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 2,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 35,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.88,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132594,
      "usage": {
        "input_tokens": 3157,
        "output_tokens": 20,
        "cost": 0.000132594
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.88
          }
        },
        "usage": {
          "input_tokens": 3157,
          "output_tokens": 20,
          "cost": 0.000132594
        },
        "id": "gen-dec-1790931017-EseXeQTSTWJOD6EN0ShO",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931017-EseXeQTSTWJOD6EN0ShO",
      "answer": {
        "type": "noul",
        "noul": 0.88
      },
      "latency_s": 0.26477379200514406,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 387
    },
    {
      "case_id": "cache-maintenance-last-day",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 24,
      "started_at": "2026-10-02T08:50:17.482209+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 2, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"thermal\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 35, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 1, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002991,
      "usage": {
        "prompt_tokens": 2891,
        "completion_tokens": 20,
        "total_tokens": 2911,
        "cost": 0.0002991,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002991,
          "upstream_inference_prompt_cost": 0.0002891,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931017-6FBbOVqJJnn76NZ2ecBy",
        "object": "chat.completion",
        "created": 1790931017,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2891,
          "completion_tokens": 20,
          "total_tokens": 2911,
          "cost": 0.0002991,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002991,
            "upstream_inference_prompt_cost": 0.0002891,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931017-6FBbOVqJJnn76NZ2ecBy",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.249428291019285,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 388
    },
    {
      "case_id": "cache-induction-boundary",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 25,
      "started_at": "2026-10-02T08:50:18.732374+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"optical\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 365, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 8, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 180, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931018-DFzf0YvXDPjtmG7XDcoF",
        "object": "chat.completion",
        "created": 1790931018,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931018-DFzf0YvXDPjtmG7XDcoF",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.257513332995586,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 389
    },
    {
      "case_id": "cache-induction-boundary",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 26,
      "started_at": "2026-10-02T08:50:19.990604+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"optical\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 365, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 8, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 180, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931020-wN3JtSkcUDQxemKhMDDo",
        "object": "chat.completion",
        "created": 1790931020,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931020-wN3JtSkcUDQxemKhMDDo",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.8913753750093747,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 390
    },
    {
      "case_id": "cache-induction-boundary",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 27,
      "started_at": "2026-10-02T08:50:20.882571+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "optical",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 365,
            "category_endorsed": true,
            "endorsement_age_days": 180,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 8,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.95,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132762,
      "usage": {
        "input_tokens": 3161,
        "output_tokens": 20,
        "cost": 0.000132762
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.95
          }
        },
        "usage": {
          "input_tokens": 3161,
          "output_tokens": 20,
          "cost": 0.000132762
        },
        "id": "gen-dec-1790931020-nR6WYxswDLYPwgljdlxq",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931020-nR6WYxswDLYPwgljdlxq",
      "answer": {
        "type": "noul",
        "noul": 0.95
      },
      "latency_s": 0.34189279199927114,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 391
    },
    {
      "case_id": "cache-sterile-broken-seal",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 28,
      "started_at": "2026-10-02T08:50:21.225136+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sterile\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": false, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931021-Mo6ottcKEKR9KyEVU6OD",
        "object": "chat.completion",
        "created": 1790931021,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931021-Mo6ottcKEKR9KyEVU6OD",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.8773167089966591,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 392
    },
    {
      "case_id": "cache-sterile-broken-seal",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 29,
      "started_at": "2026-10-02T08:50:22.103030+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "sterile",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": true,
            "duty_manager_signed": true,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": false,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.04,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132678,
      "usage": {
        "input_tokens": 3159,
        "output_tokens": 20,
        "cost": 0.000132678
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.04
          }
        },
        "usage": {
          "input_tokens": 3159,
          "output_tokens": 20,
          "cost": 0.000132678
        },
        "id": "gen-dec-1790931022-TEP2vI61UE2EtGuhNWYP",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931022-TEP2vI61UE2EtGuhNWYP",
      "answer": {
        "type": "noul",
        "noul": 0.04
      },
      "latency_s": 0.242957208014559,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 393
    },
    {
      "case_id": "cache-sterile-broken-seal",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 30,
      "started_at": "2026-10-02T08:50:22.346840+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sterile\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": false, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931022-Or85FhNA8cCUrUHj6nM2",
        "object": "chat.completion",
        "created": 1790931022,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931022-Or85FhNA8cCUrUHj6nM2",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.2426882919971831,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 394
    },
    {
      "case_id": "cache-sensor-indoor-weather",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 31,
      "started_at": "2026-10-02T08:50:23.590122+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "sensor",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": false,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 2,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 20
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.94,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132594,
      "usage": {
        "input_tokens": 3157,
        "output_tokens": 20,
        "cost": 0.000132594
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.94
          }
        },
        "usage": {
          "input_tokens": 3157,
          "output_tokens": 20,
          "cost": 0.000132594
        },
        "id": "gen-dec-1790931023-A6OB1JUQye8wqwCb54c5",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931023-A6OB1JUQye8wqwCb54c5",
      "answer": {
        "type": "noul",
        "noul": 0.94
      },
      "latency_s": 0.24653745800605975,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 395
    },
    {
      "case_id": "cache-sensor-indoor-weather",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 32,
      "started_at": "2026-10-02T08:50:23.837351+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 2, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 20, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": false}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.123e-05,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 7.123e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.123e-05,
          "upstream_inference_prompt_cost": 6.123e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931023-eD0lnPGKT7OqQuXhxN9s",
        "object": "chat.completion",
        "created": 1790931023,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 7.123e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.123e-05,
            "upstream_inference_prompt_cost": 6.123e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931023-eD0lnPGKT7OqQuXhxN9s",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9462898749916349,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 396
    },
    {
      "case_id": "cache-sensor-indoor-weather",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 33,
      "started_at": "2026-10-02T08:50:24.784350+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 2, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 20, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": false}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002991,
      "usage": {
        "prompt_tokens": 2891,
        "completion_tokens": 20,
        "total_tokens": 2911,
        "cost": 0.0002991,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002991,
          "upstream_inference_prompt_cost": 0.0002891,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931024-XOIwelO3ivpvN2TiGbjo",
        "object": "chat.completion",
        "created": 1790931024,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2891,
          "completion_tokens": 20,
          "total_tokens": 2911,
          "cost": 0.0002991,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002991,
            "upstream_inference_prompt_cost": 0.0002891,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931024-XOIwelO3ivpvN2TiGbjo",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9403079170151614,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 397
    },
    {
      "case_id": "cache-sensor-outdoor-weather",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 34,
      "started_at": "2026-10-02T08:50:25.725220+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "sensor",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": false,
            "weather_clear": false,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.58,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132636,
      "usage": {
        "input_tokens": 3158,
        "output_tokens": 20,
        "cost": 0.000132636
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.58
          }
        },
        "usage": {
          "input_tokens": 3158,
          "output_tokens": 20,
          "cost": 0.000132636
        },
        "id": "gen-dec-1790931025-YMFphLaT7HUDxXVL8Ajo",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931025-YMFphLaT7HUDxXVL8Ajo",
      "answer": {
        "type": "noul",
        "noul": 0.58
      },
      "latency_s": 0.3069650410034228,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 398
    },
    {
      "case_id": "cache-sensor-outdoor-weather",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 35,
      "started_at": "2026-10-02T08:50:26.032798+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": false, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": false}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.123e-05,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 7.123e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.123e-05,
          "upstream_inference_prompt_cost": 6.123e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931026-IFBxgqPGLDKgksNCpfrg",
        "object": "chat.completion",
        "created": 1790931026,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 7.123e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.123e-05,
            "upstream_inference_prompt_cost": 6.123e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931026-IFBxgqPGLDKgksNCpfrg",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.7980340000067372,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 399
    },
    {
      "case_id": "cache-sensor-outdoor-weather",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 36,
      "started_at": "2026-10-02T08:50:26.831558+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": false, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": false}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002991,
      "usage": {
        "prompt_tokens": 2891,
        "completion_tokens": 20,
        "total_tokens": 2911,
        "cost": 0.0002991,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002991,
          "upstream_inference_prompt_cost": 0.0002891,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931026-pAXGzcKogacW899DTgm4",
        "object": "chat.completion",
        "created": 1790931026,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2891,
          "completion_tokens": 20,
          "total_tokens": 2911,
          "cost": 0.0002991,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002991,
            "upstream_inference_prompt_cost": 0.0002891,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931026-pAXGzcKogacW899DTgm4",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.8825765829824377,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 400
    },
    {
      "case_id": "cache-after-hours-unsigned",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 37,
      "started_at": "2026-10-02T08:50:27.714991+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": true, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"rotating\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": false, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931027-2gRBKEIVyvtEkD8dVZ7s",
        "object": "chat.completion",
        "created": 1790931027,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931027-2gRBKEIVyvtEkD8dVZ7s",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.0772470829833765,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 401
    },
    {
      "case_id": "cache-after-hours-unsigned",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 38,
      "started_at": "2026-10-02T08:50:28.792894+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "rotating",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": false,
            "after_hours_permit": true,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.71,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132678,
      "usage": {
        "input_tokens": 3159,
        "output_tokens": 20,
        "cost": 0.000132678
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.71
          }
        },
        "usage": {
          "input_tokens": 3159,
          "output_tokens": 20,
          "cost": 0.000132678
        },
        "id": "gen-dec-1790931028-BFfNPg9k1I2YQahLtJrU",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931028-BFfNPg9k1I2YQahLtJrU",
      "answer": {
        "type": "noul",
        "noul": 0.71
      },
      "latency_s": 0.24040304101072252,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 402
    },
    {
      "case_id": "cache-after-hours-unsigned",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 39,
      "started_at": "2026-10-02T08:50:29.033977+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": true, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"rotating\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": false, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931029-IiHQKVYdBTxeTIlk2ukC",
        "object": "chat.completion",
        "created": 1790931029,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931029-IiHQKVYdBTxeTIlk2ukC",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.021681041980628,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 403
    },
    {
      "case_id": "cache-rotating-limits",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 40,
      "started_at": "2026-10-02T08:50:30.056489+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 0, \"calibration_current\": true, \"category\": \"rotating\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 2, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 5, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 100, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931030-SiecRxSCoIQIFidmEXId",
        "object": "chat.completion",
        "created": 1790931030,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931030-SiecRxSCoIQIFidmEXId",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.8910249159962405,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 404
    },
    {
      "case_id": "cache-rotating-limits",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 41,
      "started_at": "2026-10-02T08:50:30.948083+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "rotating",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 100,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 2,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 5,
            "battery_percent": 0
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.73,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132594,
      "usage": {
        "input_tokens": 3157,
        "output_tokens": 20,
        "cost": 0.000132594
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.73
          }
        },
        "usage": {
          "input_tokens": 3157,
          "output_tokens": 20,
          "cost": 0.000132594
        },
        "id": "gen-dec-1790931030-m9d4eink24HmzsprCYIp",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931030-m9d4eink24HmzsprCYIp",
      "answer": {
        "type": "noul",
        "noul": 0.73
      },
      "latency_s": 0.25438679099897854,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 405
    },
    {
      "case_id": "cache-rotating-limits",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 42,
      "started_at": "2026-10-02T08:50:31.203124+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 0, \"calibration_current\": true, \"category\": \"rotating\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 2, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 5, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 100, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931031-NU4PMSthFYB9bm56v7Rc",
        "object": "chat.completion",
        "created": 1790931031,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931031-NU4PMSthFYB9bm56v7Rc",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9619238339946605,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 406
    },
    {
      "case_id": "cache-rotating-no-guard",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 43,
      "started_at": "2026-10-02T08:50:32.165629+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "rotating",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": true,
            "duty_manager_signed": true,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": false,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.06,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132678,
      "usage": {
        "input_tokens": 3159,
        "output_tokens": 20,
        "cost": 0.000132678
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.06
          }
        },
        "usage": {
          "input_tokens": 3159,
          "output_tokens": 20,
          "cost": 0.000132678
        },
        "id": "gen-dec-1790931032-Rk2VsQhYdul3JCqJZzy7",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931032-Rk2VsQhYdul3JCqJZzy7",
      "answer": {
        "type": "noul",
        "noul": 0.06
      },
      "latency_s": 0.2542507080070209,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 407
    },
    {
      "case_id": "cache-rotating-no-guard",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 44,
      "started_at": "2026-10-02T08:50:32.420574+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"rotating\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": false, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931032-tk7tp9Fy84SeLPyqmqdU",
        "object": "chat.completion",
        "created": 1790931032,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931032-tk7tp9Fy84SeLPyqmqdU",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.130829374975292,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 408
    },
    {
      "case_id": "cache-rotating-no-guard",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 45,
      "started_at": "2026-10-02T08:50:33.552135+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"rotating\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": false, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931033-pTNPUcO31MsIru59Dtbh",
        "object": "chat.completion",
        "created": 1790931033,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931033-pTNPUcO31MsIru59Dtbh",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9242535840021446,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 409
    },
    {
      "case_id": "cache-trainee-duration-limit",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 46,
      "started_at": "2026-10-02T08:50:34.477194+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "sensor",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "trainee",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": true,
            "supervisor_present": true,
            "supervisor_category_endorsed": true,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 4.1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.59,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132762,
      "usage": {
        "input_tokens": 3161,
        "output_tokens": 20,
        "cost": 0.000132762
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.59
          }
        },
        "usage": {
          "input_tokens": 3161,
          "output_tokens": 20,
          "cost": 0.000132762
        },
        "id": "gen-dec-1790931034-xXbv7XVqf5O3RTlUQNuN",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931034-xXbv7XVqf5O3RTlUQNuN",
      "answer": {
        "type": "noul",
        "noul": 0.59
      },
      "latency_s": 0.2415909169940278,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 410
    },
    {
      "case_id": "cache-trainee-duration-limit",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 47,
      "started_at": "2026-10-02T08:50:34.719659+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 4.1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"trainee\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": true, \"supervisor_category_endorsed\": true, \"supervisor_present\": true, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002994,
      "usage": {
        "prompt_tokens": 2894,
        "completion_tokens": 20,
        "total_tokens": 2914,
        "cost": 0.0002994,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002994,
          "upstream_inference_prompt_cost": 0.0002894,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931034-hjk64JfBCqyWx0hw0NoQ",
        "object": "chat.completion",
        "created": 1790931034,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2894,
          "completion_tokens": 20,
          "total_tokens": 2914,
          "cost": 0.0002994,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002994,
            "upstream_inference_prompt_cost": 0.0002894,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931034-hjk64JfBCqyWx0hw0NoQ",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 2.8019795840082224,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 411
    },
    {
      "case_id": "cache-trainee-duration-limit",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 48,
      "started_at": "2026-10-02T08:50:37.522490+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 4.1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"trainee\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": true, \"supervisor_category_endorsed\": true, \"supervisor_present\": true, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.153e-05,
      "usage": {
        "prompt_tokens": 2895,
        "completion_tokens": 20,
        "total_tokens": 2915,
        "cost": 7.153e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.153e-05,
          "upstream_inference_prompt_cost": 6.153e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931037-pX8agRRtCE47QK5OPZcf",
        "object": "chat.completion",
        "created": 1790931037,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2895,
          "completion_tokens": 20,
          "total_tokens": 2915,
          "cost": 7.153e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.153e-05,
            "upstream_inference_prompt_cost": 6.153e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931037-pX8agRRtCE47QK5OPZcf",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.8757535830081906,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 412
    },
    {
      "case_id": "cache-trainee-supervised",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 49,
      "started_at": "2026-10-02T08:50:38.399759+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "portable",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "trainee",
            "days_since_induction": 30,
            "category_endorsed": false,
            "endorsement_age_days": 30,
            "supervisor_active": true,
            "supervisor_present": true,
            "supervisor_category_endorsed": true,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 4,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.89,
      "valid": true,
      "error": null,
      "cost_usd": 0.00013272,
      "usage": {
        "input_tokens": 3160,
        "output_tokens": 20,
        "cost": 0.00013272
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.89
          }
        },
        "usage": {
          "input_tokens": 3160,
          "output_tokens": 20,
          "cost": 0.00013272
        },
        "id": "gen-dec-1790931038-pgonJeipl3Io6w5jpM3J",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931038-pgonJeipl3Io6w5jpM3J",
      "answer": {
        "type": "noul",
        "noul": 0.89
      },
      "latency_s": 0.24228216698975302,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 413
    },
    {
      "case_id": "cache-trainee-supervised",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 50,
      "started_at": "2026-10-02T08:50:38.642801+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"portable\", \"category_endorsed\": false, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 4, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"trainee\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": true, \"supervisor_category_endorsed\": true, \"supervisor_present\": true, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931038-tJG8lufkgDOng840Jct7",
        "object": "chat.completion",
        "created": 1790931038,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931038-tJG8lufkgDOng840Jct7",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0676087499887217,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 414
    },
    {
      "case_id": "cache-trainee-supervised",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 51,
      "started_at": "2026-10-02T08:50:39.711302+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"portable\", \"category_endorsed\": false, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 4, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"trainee\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": true, \"supervisor_category_endorsed\": true, \"supervisor_present\": true, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931039-aRMUtmeSkFPsFiUndyCw",
        "object": "chat.completion",
        "created": 1790931039,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931039-aRMUtmeSkFPsFiUndyCw",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.8322054999880493,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 415
    },
    {
      "case_id": "cache-maintenance-due",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 52,
      "started_at": "2026-10-02T08:50:40.544283+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"thermal\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 0, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002991,
      "usage": {
        "prompt_tokens": 2891,
        "completion_tokens": 20,
        "total_tokens": 2911,
        "cost": 0.0002991,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002991,
          "upstream_inference_prompt_cost": 0.0002891,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931040-BrZpQPrRoszdBqSi0Xau",
        "object": "chat.completion",
        "created": 1790931040,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2891,
          "completion_tokens": 20,
          "total_tokens": 2911,
          "cost": 0.0002991,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002991,
            "upstream_inference_prompt_cost": 0.0002891,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931040-BrZpQPrRoszdBqSi0Xau",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9625105420127511,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 416
    },
    {
      "case_id": "cache-maintenance-due",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 53,
      "started_at": "2026-10-02T08:50:41.507510+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"thermal\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 0, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.123e-05,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 7.123e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.123e-05,
          "upstream_inference_prompt_cost": 6.123e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931041-oxEMVskOYlRPR1DtK5la",
        "object": "chat.completion",
        "created": 1790931041,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 7.123e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.123e-05,
            "upstream_inference_prompt_cost": 6.123e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931041-oxEMVskOYlRPR1DtK5la",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.1144889169954695,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 417
    },
    {
      "case_id": "cache-maintenance-due",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 54,
      "started_at": "2026-10-02T08:50:42.622670+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "thermal",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 0,
            "release_tag": "green",
            "emergency": true,
            "duty_manager_signed": true,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132594,
      "usage": {
        "input_tokens": 3157,
        "output_tokens": 20,
        "cost": 0.000132594
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 3157,
          "output_tokens": 20,
          "cost": 0.000132594
        },
        "id": "gen-dec-1790931042-IsXmDFUgI8oIXNyPX83X",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931042-IsXmDFUgI8oIXNyPX83X",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.30418325000209734,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 418
    },
    {
      "case_id": "cache-sterile-limits",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 55,
      "started_at": "2026-10-02T08:50:42.927523+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 10, \"calibration_current\": true, \"category\": \"sterile\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 6, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 25, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 300, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931042-bmfGnb3VqEBl2FcJUmjj",
        "object": "chat.completion",
        "created": 1790931042,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931042-bmfGnb3VqEBl2FcJUmjj",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.1758441670099273,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 419
    },
    {
      "case_id": "cache-sterile-limits",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 56,
      "started_at": "2026-10-02T08:50:44.104041+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "sterile",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 300,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 6,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 25,
            "battery_percent": 10
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.79,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132678,
      "usage": {
        "input_tokens": 3159,
        "output_tokens": 20,
        "cost": 0.000132678
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.79
          }
        },
        "usage": {
          "input_tokens": 3159,
          "output_tokens": 20,
          "cost": 0.000132678
        },
        "id": "gen-dec-1790931044-OqIfhT2MFLdz9RhVL2nG",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931044-OqIfhT2MFLdz9RhVL2nG",
      "answer": {
        "type": "noul",
        "noul": 0.79
      },
      "latency_s": 0.24528437497792765,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 420
    },
    {
      "case_id": "cache-sterile-limits",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 57,
      "started_at": "2026-10-02T08:50:44.349967+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 10, \"calibration_current\": true, \"category\": \"sterile\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 6, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 25, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 300, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931044-NkpnK2PlW9cMHQTB3kSo",
        "object": "chat.completion",
        "created": 1790931044,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931044-NkpnK2PlW9cMHQTB3kSo",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9116793329885695,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 421
    },
    {
      "case_id": "cache-portable-battery-low",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 58,
      "started_at": "2026-10-02T08:50:45.262469+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 29.9, \"calibration_current\": true, \"category\": \"portable\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.143e-05,
      "usage": {
        "prompt_tokens": 2894,
        "completion_tokens": 20,
        "total_tokens": 2914,
        "cost": 7.143e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.143e-05,
          "upstream_inference_prompt_cost": 6.143e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931045-s8Fw2zGZXTDVAv4dojSB",
        "object": "chat.completion",
        "created": 1790931045,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2894,
          "completion_tokens": 20,
          "total_tokens": 2914,
          "cost": 7.143e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.143e-05,
            "upstream_inference_prompt_cost": 6.143e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931045-s8Fw2zGZXTDVAv4dojSB",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.8947590420139022,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 422
    },
    {
      "case_id": "cache-portable-battery-low",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 59,
      "started_at": "2026-10-02T08:50:46.157859+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "portable",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": true,
            "duty_manager_signed": true,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 29.9
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.73,
      "valid": true,
      "error": null,
      "cost_usd": 0.00013272,
      "usage": {
        "input_tokens": 3160,
        "output_tokens": 20,
        "cost": 0.00013272
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.73
          }
        },
        "usage": {
          "input_tokens": 3160,
          "output_tokens": 20,
          "cost": 0.00013272
        },
        "id": "gen-dec-1790931046-4CmUJ1HATp6rGi987XXp",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931046-4CmUJ1HATp6rGi987XXp",
      "answer": {
        "type": "noul",
        "noul": 0.73
      },
      "latency_s": 0.3111065409902949,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 423
    },
    {
      "case_id": "cache-portable-battery-low",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 60,
      "started_at": "2026-10-02T08:50:46.469765+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 29.9, \"calibration_current\": true, \"category\": \"portable\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002993,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 0.0002993,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002993,
          "upstream_inference_prompt_cost": 0.0002893,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931046-YVv0GxtOBisakybZHThk",
        "object": "chat.completion",
        "created": 1790931046,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 0.0002993,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002993,
            "upstream_inference_prompt_cost": 0.0002893,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931046-YVv0GxtOBisakybZHThk",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.9271121250058059,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 424
    },
    {
      "case_id": "cache-optical-dirty-route",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 61,
      "started_at": "2026-10-02T08:50:47.398127+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"optical\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": false, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 1, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.133e-05,
      "usage": {
        "prompt_tokens": 2893,
        "completion_tokens": 20,
        "total_tokens": 2913,
        "cost": 7.133e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.133e-05,
          "upstream_inference_prompt_cost": 6.133e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931047-lUyMANMkcrZNAc0Q6xAi",
        "object": "chat.completion",
        "created": 1790931047,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2893,
          "completion_tokens": 20,
          "total_tokens": 2913,
          "cost": 7.133e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.133e-05,
            "upstream_inference_prompt_cost": 6.133e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931047-lUyMANMkcrZNAc0Q6xAi",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9441170420031995,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 425
    },
    {
      "case_id": "cache-optical-dirty-route",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 62,
      "started_at": "2026-10-02T08:50:48.343063+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"optical\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": false, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": false, \"emergency\": false, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 1, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002992,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 0.0002992,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002992,
          "upstream_inference_prompt_cost": 0.0002892,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931048-S1eNC28YM1YpOuy7hiFp",
        "object": "chat.completion",
        "created": 1790931048,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 0.0002992,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002992,
            "upstream_inference_prompt_cost": 0.0002892,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931048-S1eNC28YM1YpOuy7hiFp",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9778748340031598,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 426
    },
    {
      "case_id": "cache-optical-dirty-route",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 63,
      "started_at": "2026-10-02T08:50:49.321691+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "optical",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": false,
            "duty_manager_signed": false,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 1,
            "protective_case": true,
            "clean_route": false,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.09,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132636,
      "usage": {
        "input_tokens": 3158,
        "output_tokens": 20,
        "cost": 0.000132636
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.09
          }
        },
        "usage": {
          "input_tokens": 3158,
          "output_tokens": 20,
          "cost": 0.000132636
        },
        "id": "gen-dec-1790931049-JrhaSd8M1ym9M7hTWxMi",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931049-JrhaSd8M1ym9M7hTWxMi",
      "answer": {
        "type": "noul",
        "noul": 0.09
      },
      "latency_s": 0.25537312499363907,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 427
    },
    {
      "case_id": "cache-amber-exception",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 64,
      "started_at": "2026-10-02T08:50:49.577786+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"thermal\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 0, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"amber\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.123e-05,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 7.123e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.123e-05,
          "upstream_inference_prompt_cost": 6.123e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931049-jGtzcJBJiTTs4x6PhGNr",
        "object": "chat.completion",
        "created": 1790931049,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 7.123e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.123e-05,
            "upstream_inference_prompt_cost": 6.123e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931049-jGtzcJBJiTTs4x6PhGNr",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 0.8403459999826737,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 428
    },
    {
      "case_id": "cache-amber-exception",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 65,
      "started_at": "2026-10-02T08:50:50.418900+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "thermal",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "amber",
            "emergency": true,
            "duty_manager_signed": true,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 0,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.85,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132636,
      "usage": {
        "input_tokens": 3158,
        "output_tokens": 20,
        "cost": 0.000132636
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.85
          }
        },
        "usage": {
          "input_tokens": 3158,
          "output_tokens": 20,
          "cost": 0.000132636
        },
        "id": "gen-dec-1790931050-hO7c5QAKVB8zNbybN3VF",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931050-hO7c5QAKVB8zNbybN3VF",
      "answer": {
        "type": "noul",
        "noul": 0.85
      },
      "latency_s": 0.2529080000240356,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 429
    },
    {
      "case_id": "cache-amber-exception",
      "experiment": "cache_policy",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 66,
      "started_at": "2026-10-02T08:50:50.672547+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"thermal\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 0, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"amber\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002991,
      "usage": {
        "prompt_tokens": 2891,
        "completion_tokens": 20,
        "total_tokens": 2911,
        "cost": 0.0002991,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002991,
          "upstream_inference_prompt_cost": 0.0002891,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931050-LroAF4GptaAyCBZ9cpw4",
        "object": "chat.completion",
        "created": 1790931050,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2891,
          "completion_tokens": 20,
          "total_tokens": 2911,
          "cost": 0.0002991,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002991,
            "upstream_inference_prompt_cost": 0.0002891,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931050-LroAF4GptaAyCBZ9cpw4",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.0964420000091195,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 430
    },
    {
      "case_id": "cache-reservation-no-waiver",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 67,
      "started_at": "2026-10-02T08:50:51.769718+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 1, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002991,
      "usage": {
        "prompt_tokens": 2891,
        "completion_tokens": 20,
        "total_tokens": 2911,
        "cost": 0.0002991,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002991,
          "upstream_inference_prompt_cost": 0.0002891,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931051-QY66aIwCvrr8jiKbp9pC",
        "object": "chat.completion",
        "created": 1790931051,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2891,
          "completion_tokens": 20,
          "total_tokens": 2911,
          "cost": 0.0002991,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002991,
            "upstream_inference_prompt_cost": 0.0002891,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931051-QY66aIwCvrr8jiKbp9pC",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.297327000007499,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 431
    },
    {
      "case_id": "cache-reservation-no-waiver",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 68,
      "started_at": "2026-10-02T08:50:53.067948+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"sensor\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"green\", \"requester_role\": \"member\", \"reservation_conflicts\": 1, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.123e-05,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 7.123e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.123e-05,
          "upstream_inference_prompt_cost": 6.123e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931053-Z81B8a0jKsxyQoLFwVfV",
        "object": "chat.completion",
        "created": 1790931053,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 7.123e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.123e-05,
            "upstream_inference_prompt_cost": 6.123e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931053-Z81B8a0jKsxyQoLFwVfV",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9621299999998882,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 432
    },
    {
      "case_id": "cache-reservation-no-waiver",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 69,
      "started_at": "2026-10-02T08:50:54.030683+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "sensor",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "green",
            "emergency": true,
            "duty_manager_signed": true,
            "calibration_current": true,
            "reservation_conflicts": 1,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.04,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132636,
      "usage": {
        "input_tokens": 3158,
        "output_tokens": 20,
        "cost": 0.000132636
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.04
          }
        },
        "usage": {
          "input_tokens": 3158,
          "output_tokens": 20,
          "cost": 0.000132636
        },
        "id": "gen-dec-1790931054-uP67e0iYMhiOhhtGj24B",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931054-uP67e0iYMhiOhhtGj24B",
      "answer": {
        "type": "noul",
        "noul": 0.04
      },
      "latency_s": 0.24389537499519065,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 433
    },
    {
      "case_id": "cache-red-no-exception",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna cached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 70,
      "started_at": "2026-10-02T08:50:54.275258+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\",",
                "prompt_cache_breakpoint": {
                  "mode": "explicit"
                }
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"thermal\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"red\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 7.123e-05,
      "usage": {
        "prompt_tokens": 2892,
        "completion_tokens": 20,
        "total_tokens": 2912,
        "cost": 7.123e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 2533,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.123e-05,
          "upstream_inference_prompt_cost": 6.123e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931054-F6mwcGXnPKASuWV2QCPS",
        "object": "chat.completion",
        "created": 1790931054,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2892,
          "completion_tokens": 20,
          "total_tokens": 2912,
          "cost": 7.123e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 2533,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 7.123e-05,
            "upstream_inference_prompt_cost": 6.123e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931054-F6mwcGXnPKASuWV2QCPS",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.9387980000174139,
      "cached_tokens": 2533,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 434
    },
    {
      "case_id": "cache-red-no-exception",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna uncached",
      "requested_model": "openai/gpt-6-luna",
      "phase": "measured",
      "sequence": 71,
      "started_at": "2026-10-02T08:50:55.214846+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": [
              {
                "type": "text",
                "text": "{\"state\":{\"rulebook\":\"NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\\n\\n1. Scope and decision\\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\\n\\n2. Equipment catalogue\\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\\n\\n3. Interpretation and precedence\\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\\n\\n4. Requester and access status\\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\\n\\n5. General safety induction\\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\\n\\n6. Individual category endorsement\\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\\n\\n7. Supervised checkout\\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\\n\\n8. Asset condition and service status\\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\\n\\n9. Calibration and reservations\\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\\n\\n10. Pickup timing and advance notice\\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\\n\\n11. Approved place of use\\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\\n\\n12. Transport route\\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\\n\\n13. Duration and workload\\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\\n\\n14. Electrical, thermal, and rotating operation\\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\\n\\n15. Sterile handling\\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\\n\\n16. Environment and battery\\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\\n\\n17. Record completeness and the final answer\\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\\n\","
              },
              {
                "type": "text",
                "text": "\"facts\":{\"active_other_loans\": 0, \"after_hours_permit\": false, \"area_exclusion\": true, \"badge_active\": true, \"battery_percent\": 100, \"calibration_current\": true, \"category\": \"thermal\", \"category_endorsed\": true, \"clean_destination\": true, \"clean_route\": true, \"days_since_induction\": 30, \"destination_approved\": true, \"destination_indoor\": true, \"duration_hours\": 1, \"duty_manager_signed\": true, \"emergency\": true, \"endorsement_age_days\": 30, \"environment_temperature_c\": 20, \"fire_watch\": true, \"guard_fitted\": true, \"heat_shield\": true, \"known_damage\": false, \"maintenance_days_remaining\": 10, \"membership_active\": true, \"notice_hours\": 4, \"pickup_staffed\": true, \"power_supply_tested\": true, \"protective_case\": true, \"release_tag\": \"red\", \"requester_role\": \"member\", \"reservation_conflicts\": 0, \"seal_intact\": true, \"sterile_witness\": true, \"supervisor_active\": false, \"supervisor_category_endorsed\": false, \"supervisor_present\": false, \"transport_distance_m\": 50, \"unresolved_suspension\": false, \"vibration_isolation\": true, \"weather_clear\": true}},\"questions\":{\"answer\": {\"criteria\": {\"false\": \"At least one applicable requirement fails.\", \"true\": \"Every applicable requirement passes, including any precisely stated exceptions.\"}, \"instructions\": \"Must this equipment dispatch be approved now under the supplied fictional rulebook?\", \"type\": \"noul\"}}}"
              }
            ]
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "stream": false,
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "prompt_cache_options": {
          "mode": "explicit",
          "ttl": "30m"
        },
        "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051"
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 0.0002991,
      "usage": {
        "prompt_tokens": 2891,
        "completion_tokens": 20,
        "total_tokens": 2911,
        "cost": 0.0002991,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0002991,
          "upstream_inference_prompt_cost": 0.0002891,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790931055-yoRibzJ0QhJbF9CSSmoF",
        "object": "chat.completion",
        "created": 1790931055,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 2891,
          "completion_tokens": 20,
          "total_tokens": 2911,
          "cost": 0.0002991,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 0.0002991,
            "upstream_inference_prompt_cost": 0.0002891,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790931055-yoRibzJ0QhJbF9CSSmoF",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 1.019431833003182,
      "cached_tokens": 0,
      "cache_write_tokens": 0,
      "run": "cache",
      "record_index": 435
    },
    {
      "case_id": "cache-red-no-exception",
      "experiment": "cache_policy",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "measured",
      "sequence": 72,
      "started_at": "2026-10-02T08:50:56.234904+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rulebook": "NORTH QUAY SHARED EQUIPMENT SERVICE — DISPATCH RULEBOOK, VERSION 1\n\n1. Scope and decision\nThis fictional service lends six categories of equipment inside a research campus. Decide whether the service must approve the requested dispatch now. Approval means every applicable requirement below passes. A request that fails any requirement is not approved, even if it would pass most others. This is a deterministic policy decision, not an estimate of future behaviour. All facts in a request are authoritative for this exercise. Do not infer missing permission from a persuasive purpose or a low level of apparent risk. Apply only the written rules, using the supplied numerical facts rather than real-world assumptions about equipment, employment, or safety. A fact marked false is false, rather than merely undocumented.\n\n2. Equipment catalogue\nThere are exactly six categories: optical, thermal, rotating, portable, sensor, and sterile. Their maximum ordinary checkout durations, in hours, are respectively 8, 4, 2, 24, 72, and 6. Their minimum remaining battery percentages are respectively 0, 0, 0, 30, 20, and 10. Zero means there is no battery requirement for that category; it does not mean the asset must have an empty battery. The standard maximum transport distance is 400 metres for optical, 200 for thermal, 100 for rotating, 2000 for portable, 3000 for sensor, and 300 for sterile. Catalogue limits are inclusive. A duration equal to its category limit passes the duration rule. A transport distance equal to its applicable maximum passes the transport rule.\n\n3. Interpretation and precedence\nEvery numbered section is cumulative except for an exception explicitly named in that section. An exception changes only the particular condition it names. A duty-manager signature has no general overriding force. Likewise an emergency flag is not a universal waiver. If an exception applies, continue checking the other requirements. The words 'and' and 'both' mean every listed condition must hold; 'or' means at least one listed condition must hold. Distances, hours, days, temperatures, counts, and percentages are ordinary numeric values. Do not round a value before comparison. A value just above a maximum fails even if the difference is small. Descriptive notes never override structured facts.\n\n4. Requester and access status\nThe requester must have an active membership, an active access badge, and no unresolved equipment suspension. These are three independent conditions. A supervisor cannot cure an expired badge or a suspended membership. A duty manager also cannot waive these conditions. The service distinguishes trained members, trainees, and visitors, using the requester_role field. A visitor is never eligible for checkout under this rulebook, regardless of supervision. A trainee can be eligible through the supervision process in section 7. An ordinary member can be eligible through an individual endorsement or through that same supervision process. For the purpose of this test, these are the only requester roles.\n\n5. General safety induction\nAn otherwise eligible requester must have completed the campus safety induction at most 365 days before dispatch. The days_since_induction field is the elapsed number of full days since that completion. Day zero means completion today. Day 365 still passes; day 366 fails. A negative number is invalid and fails. Category endorsement, experience, and an attending supervisor do not replace the general induction. This condition remains in force during an emergency. The rule is intentionally based on elapsed days rather than calendar years, so leap years and dates need not be considered. Each case contains the elapsed value directly, and that value must be used without recalculation.\n\n6. Individual category endorsement\nAn ordinary member may satisfy the category qualification requirement by holding a category endorsement completed at most 180 days ago, provided category_endorsed is true. Both the endorsement flag and its age must pass. An age from zero through 180 days is acceptable. An expired endorsement with age 181 does not qualify, even if the requester has often used the device. Trainees cannot qualify through their own endorsement alone: they must meet the supervision process. A member whose endorsement is absent or expired may also meet the supervision process instead. This alternative affects the category qualification condition only; it does not waive induction, access, asset condition, or any operational requirement.\n\n7. Supervised checkout\nSupervision qualifies a requester only when supervisor_active, supervisor_present, and supervisor_category_endorsed are all true. These facts mean, respectively, that the supervisor is currently authorised, will be physically present throughout use, and currently holds the category-specific endorsement. Remote availability by telephone does not satisfy physical presence. A supervisor who is present but lacks the appropriate endorsement also fails. Trainees must use this route, while ordinary members may choose this route when their individual endorsement does not qualify. A qualifying supervisor does not change the maximum duration, transport distance, battery limit, or location rules. The supervision requirement is evaluated independently from the separate sterile handling witness in section 15.\n\n8. Asset condition and service status\nThe asset must have no known damage and must have at least one full day remaining until its next required maintenance. maintenance_days_remaining equal to one passes; zero fails. Negative values fail as overdue. The release tag is green, amber, or red. A green tag passes the tag condition. A red tag always fails, including during an emergency. An amber tag passes only when the request is an emergency and a duty manager has signed it. Both flags are required for that amber exception. The exception does not repair damage or waive maintenance. For example, an amber asset with the necessary signatures still fails if known_damage is true.\n\n9. Calibration and reservations\nOptical, thermal, rotating, and sensor devices require calibration_current to be true. Portable and sterile devices have no calibration requirement in this policy. A false calibration flag for those two categories is therefore irrelevant. Every category also requires reservation_conflicts to equal zero. One or more overlapping reservations fails the reservation condition. Manager signature, emergency use, and supervision do not override another reservation. The conflict count is already calculated from the proposed time window; do not compare actual dates. A zero count passes even when other bookings exist outside that window. The policy does not ask the model to decide which of two requesters is more deserving.\n\n10. Pickup timing and advance notice\nThe requested pickup must be during staffed opening hours, represented by pickup_staffed. When pickup_staffed is false, the timing condition passes only if after_hours_permit and duty_manager_signed are both true. The after-hours permit alone is insufficient. Independently, the request must provide at least two hours of advance notice through notice_hours. The notice minimum is waived only when emergency and duty_manager_signed are both true. A signed non-emergency request still needs two hours. A signed emergency may have zero notice, but it still needs staffed pickup or the named after-hours exception. No part of this section changes the reservation or qualification requirements.\n\n11. Approved place of use\nEvery request must have destination_approved equal to true. The destination approval is required even for a very short use. Optical, thermal, rotating, and sterile equipment must also remain indoors, so destination_indoor must be true for those categories. Portable and sensor equipment may be used indoors or outdoors. A sensor that will be used outdoors additionally requires weather_clear to be true. Portable equipment has no weather-clear condition. Indoor sensor use does not require clear weather. The emergency status has no effect on these location conditions. The supplied destination facts describe the actual proposed operating location, not merely the pickup point or the beginning of the transport route.\n\n12. Transport route\nThe proposed transport_distance_m must be nonnegative and no greater than the catalogue maximum for the category. All six categories need protective_case equal to true whenever transport_distance_m is greater than zero. A movement of zero metres does not require a protective case. Sterile and optical devices additionally require a clean_route whenever there is any transport. The other categories do not require a clean route under this rulebook. An approved destination does not imply that the transport route is clean. Supervision, emergency status, and a manager signature cannot increase a distance limit or replace a required protective case. The maximum distance concerns the complete one-way route, not straight-line distance on a map.\n\n13. Duration and workload\nThe requested duration_hours must be positive and must not exceed the catalogue limit. For trainees, the duration must also be at most four hours, even when the catalogue permits longer borrowing. Apply the smaller applicable limit. A member using supervision keeps the ordinary category duration limit and is not treated as a trainee. Every requester must have active_other_loans less than three. A requester with zero, one, or two other loans passes; a requester with three fails. The asset requested here is not included in that count. An emergency does not waive the loan-count limit. A manager signature does not extend a duration limit.\n\n14. Electrical, thermal, and rotating operation\nThermal and rotating equipment require power_supply_tested equal to true. Optical equipment requires vibration_isolation equal to true. Rotating equipment additionally requires both guard_fitted and area_exclusion to be true. Thermal equipment additionally requires heat_shield and fire_watch to be true. These category-specific requirements do not apply to portable, sensor, or sterile equipment. An unrelated false flag is not a reason for refusal. For example, fire_watch is irrelevant to an optical checkout, while vibration_isolation is irrelevant to a thermal checkout. All electrical and mechanical conditions remain in force during an emergency and when a supervisor attends. A working protective case during transport does not replace these operating protections.\n\n15. Sterile handling\nSterile equipment requires seal_intact, clean_destination, and sterile_witness to be true. All three conditions must pass. The sterile witness is a separate handling role and need not be the category supervisor. A trainee borrowing sterile equipment still needs both a qualifying supervisor and a sterile witness. The presence of one role must not be inferred to imply the other. Sterile equipment also requires at least 10 percent battery under the catalogue, regardless of whether the witness is present. A broken seal cannot be cured by a clean destination or an emergency manager signature. For non-sterile categories these three sterile-specific fields are irrelevant to approval.\n\n16. Environment and battery\nThe proposed environment_temperature_c must be between 10 and 30 degrees inclusive for optical, between 5 and 35 inclusive for thermal, between 5 and 35 inclusive for rotating, between minus 10 and 45 inclusive for portable, between minus 20 and 50 inclusive for sensor, and between 15 and 25 inclusive for sterile. A value at either endpoint passes. These are operating temperatures; they do not describe weather during transport. Compare battery_percent with the category minimum in section 2. Battery percent must also be between zero and 100 inclusive for every category. A value of 101 is invalid even if the battery minimum is zero. No signature or supervision exception changes these limits.\n\n17. Record completeness and the final answer\nThe case facts provide every field needed to apply the rules. All listed booleans and numbers are deliberate, including irrelevant fields. Do not award approval merely because the case description calls it routine, urgent, harmless, or educational. Evaluate the structured facts against all applicable sections. The target question is whether dispatch must be approved now under this rulebook, not whether it could become approvable after changes. No means at least one current requirement fails. Yes means every current requirement passes. Return the required probability object; because the policy and facts are deterministic, a completely certain application yields zero for no or one for yes. Do not add explanations or other fields.\n",
          "facts": {
            "category": "thermal",
            "membership_active": true,
            "badge_active": true,
            "unresolved_suspension": false,
            "requester_role": "member",
            "days_since_induction": 30,
            "category_endorsed": true,
            "endorsement_age_days": 30,
            "supervisor_active": false,
            "supervisor_present": false,
            "supervisor_category_endorsed": false,
            "known_damage": false,
            "maintenance_days_remaining": 10,
            "release_tag": "red",
            "emergency": true,
            "duty_manager_signed": true,
            "calibration_current": true,
            "reservation_conflicts": 0,
            "pickup_staffed": true,
            "after_hours_permit": false,
            "notice_hours": 4,
            "destination_approved": true,
            "destination_indoor": true,
            "weather_clear": true,
            "transport_distance_m": 50,
            "protective_case": true,
            "clean_route": true,
            "duration_hours": 1,
            "active_other_loans": 0,
            "power_supply_tested": true,
            "vibration_isolation": true,
            "guard_fitted": true,
            "area_exclusion": true,
            "heat_shield": true,
            "fire_watch": true,
            "seal_intact": true,
            "clean_destination": true,
            "sterile_witness": true,
            "environment_temperature_c": 20,
            "battery_percent": 100
          }
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Must this equipment dispatch be approved now under the supplied fictional rulebook?",
            "criteria": {
              "true": "Every applicable requirement passes, including any precisely stated exceptions.",
              "false": "At least one applicable requirement fails."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.07,
      "valid": true,
      "error": null,
      "cost_usd": 0.000132636,
      "usage": {
        "input_tokens": 3158,
        "output_tokens": 20,
        "cost": 0.000132636
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.07
          }
        },
        "usage": {
          "input_tokens": 3158,
          "output_tokens": 20,
          "cost": 0.000132636
        },
        "id": "gen-dec-1790931056-KE6U1P28TtyCBJQ7bJmA",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790931056-KE6U1P28TtyCBJQ7bJmA",
      "answer": {
        "type": "noul",
        "noul": 0.07
      },
      "latency_s": 0.2546017910062801,
      "cached_tokens": null,
      "cache_write_tokens": null,
      "run": "cache",
      "record_index": 436
    },
    {
      "case_id": "warmup-yes",
      "experiment": "warmup",
      "target": 1,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "warmup",
      "sequence": 0,
      "started_at": "2026-10-02T08:40:29.938428+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rule": "Approve if the number is at least 10.",
          "number": 15
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is approval required under the rule?",
            "criteria": {
              "true": "The rule requires approval.",
              "false": "The rule does not require approval."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.96,
      "valid": true,
      "error": null,
      "cost_usd": 1.3692e-05,
      "usage": {
        "input_tokens": 326,
        "output_tokens": 20,
        "cost": 1.3692e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.96
          }
        },
        "usage": {
          "input_tokens": 326,
          "output_tokens": 20,
          "cost": 1.3692e-05
        },
        "id": "gen-dec-1790930430-PnXUyR10yToLic6TzRUL",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930430-PnXUyR10yToLic6TzRUL",
      "answer": {
        "type": "noul",
        "noul": 0.96
      },
      "latency_s": 0.5134258330217563,
      "run": "setup",
      "record_index": 437
    },
    {
      "case_id": "warmup-yes",
      "experiment": "warmup",
      "target": 1,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "warmup",
      "sequence": 1,
      "started_at": "2026-10-02T08:40:30.451938+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The rule does not require approval.\", \"true\": \"The rule requires approval.\"}, \"instructions\": \"Is approval required under the rule?\", \"type\": \"noul\"}}, \"state\": {\"number\": 15, \"rule\": \"Approve if the number is at least 10.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 1.0,
      "valid": true,
      "error": null,
      "cost_usd": 3e-05,
      "usage": {
        "prompt_tokens": 200,
        "completion_tokens": 20,
        "total_tokens": 220,
        "cost": 3e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3e-05,
          "upstream_inference_prompt_cost": 2e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930430-6KHg49gHWoLE4uOHARxX",
        "object": "chat.completion",
        "created": 1790930430,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":1}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 200,
          "completion_tokens": 20,
          "total_tokens": 220,
          "cost": 3e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3e-05,
            "upstream_inference_prompt_cost": 2e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930430-6KHg49gHWoLE4uOHARxX",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 1
      },
      "latency_s": 1.2567277500056662,
      "run": "setup",
      "record_index": 438
    },
    {
      "case_id": "warmup-no",
      "experiment": "warmup",
      "target": 0,
      "target_kind": "label",
      "model_label": "Jev",
      "requested_model": "typesafe/jev-1.13",
      "phase": "warmup",
      "sequence": 2,
      "started_at": "2026-10-02T08:40:31.708754+00:00",
      "request": {
        "model": "typesafe/jev-1.13",
        "state": {
          "rule": "Approve if the number is at least 10.",
          "number": 4
        },
        "questions": {
          "answer": {
            "type": "noul",
            "instructions": "Is approval required under the rule?",
            "criteria": {
              "true": "The rule requires approval.",
              "false": "The rule does not require approval."
            }
          }
        }
      },
      "endpoint": "/api/alpha/decisions",
      "probability": 0.04,
      "valid": true,
      "error": null,
      "cost_usd": 1.365e-05,
      "usage": {
        "input_tokens": 325,
        "output_tokens": 20,
        "cost": 1.365e-05
      },
      "response_model": "typesafe/jev-1.13-20260917",
      "provider": "TypeSafe",
      "http_status": 200,
      "response": {
        "model": "typesafe/jev-1.13-20260917",
        "answers": {
          "answer": {
            "type": "noul",
            "noul": 0.04
          }
        },
        "usage": {
          "input_tokens": 325,
          "output_tokens": 20,
          "cost": 1.365e-05
        },
        "id": "gen-dec-1790930431-4Pw2DZ3Li7d5cRIejyOA",
        "provider": "TypeSafe"
      },
      "generation_id": "gen-dec-1790930431-4Pw2DZ3Li7d5cRIejyOA",
      "answer": {
        "type": "noul",
        "noul": 0.04
      },
      "latency_s": 0.22932208300335333,
      "run": "setup",
      "record_index": 439
    },
    {
      "case_id": "warmup-no",
      "experiment": "warmup",
      "target": 0,
      "target_kind": "label",
      "model_label": "Luna",
      "requested_model": "openai/gpt-6-luna",
      "phase": "warmup",
      "sequence": 3,
      "started_at": "2026-10-02T08:40:31.938167+00:00",
      "request": {
        "model": "openai/gpt-6-luna",
        "messages": [
          {
            "role": "system",
            "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."
          },
          {
            "role": "user",
            "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The rule does not require approval.\", \"true\": \"The rule requires approval.\"}, \"instructions\": \"Is approval required under the rule?\", \"type\": \"noul\"}}, \"state\": {\"number\": 4, \"rule\": \"Approve if the number is at least 10.\"}}"
          }
        ],
        "response_format": {
          "type": "json_schema",
          "json_schema": {
            "name": "noul_answer",
            "strict": true,
            "schema": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string",
                  "enum": [
                    "noul"
                  ]
                },
                "noul": {
                  "type": "number",
                  "minimum": 0,
                  "maximum": 1
                }
              },
              "required": [
                "type",
                "noul"
              ],
              "additionalProperties": false
            }
          }
        },
        "reasoning": {
          "effort": "none"
        },
        "max_tokens": 128,
        "prompt_cache_options": {
          "mode": "explicit"
        },
        "provider": {
          "only": [
            "OpenAI"
          ],
          "allow_fallbacks": false,
          "require_parameters": true
        },
        "stream": false
      },
      "endpoint": "/api/v1/chat/completions",
      "probability": 0.0,
      "valid": true,
      "error": null,
      "cost_usd": 3e-05,
      "usage": {
        "prompt_tokens": 200,
        "completion_tokens": 20,
        "total_tokens": 220,
        "cost": 3e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3e-05,
          "upstream_inference_prompt_cost": 2e-05,
          "upstream_inference_completions_cost": 1e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "response_model": "openai/gpt-6-luna",
      "provider": "OpenAI",
      "http_status": 200,
      "response": {
        "id": "gen-1790930432-iz5x6x37CsXyiDtzpFs9",
        "object": "chat.completion",
        "created": 1790930432,
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "system_fingerprint": null,
        "service_tier": "default",
        "choices": [
          {
            "index": 0,
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "completed",
            "message": {
              "role": "assistant",
              "content": "{\"type\":\"noul\",\"noul\":0}",
              "refusal": null,
              "reasoning": null
            }
          }
        ],
        "usage": {
          "prompt_tokens": 200,
          "completion_tokens": 20,
          "total_tokens": 220,
          "cost": 3e-05,
          "is_byok": false,
          "prompt_tokens_details": {
            "cached_tokens": 0,
            "cache_write_tokens": 0,
            "audio_tokens": 0,
            "video_tokens": 0
          },
          "cost_details": {
            "upstream_inference_cost": 3e-05,
            "upstream_inference_prompt_cost": 2e-05,
            "upstream_inference_completions_cost": 1e-05
          },
          "completion_tokens_details": {
            "reasoning_tokens": 0,
            "image_tokens": 0,
            "audio_tokens": 0
          }
        }
      },
      "generation_id": "gen-1790930432-iz5x6x37CsXyiDtzpFs9",
      "finish_reason": "stop",
      "answer": {
        "type": "noul",
        "noul": 0
      },
      "latency_s": 0.83921691699652,
      "run": "setup",
      "record_index": 440
    }
  ],
  "audit": [
    {
      "id": "boolq-dev-0659",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly gives the host an automatic berth."
    },
    {
      "id": "boolq-dev-2774",
      "label": "no",
      "category": "ambiguous",
      "evidence": "Source describes a proposed acquisition followed by acquisition of stores; its last sentence can be read as acquisition of Rite Aid itself. Company ownership versus store ownership and the 2017 time frame need care."
    },
    {
      "id": "boolq-dev-0579",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage says males and females are almost identical, while acknowledging male size differs."
    },
    {
      "id": "boolq-dev-2217",
      "label": "no",
      "category": "supported",
      "evidence": "SWIFT is used in addition to BSB for international transfers; they are distinct identifiers."
    },
    {
      "id": "boolq-dev-1517",
      "label": "no",
      "category": "ambiguous",
      "evidence": "Passage says the 2018 tournament took place in Russia. No is defensible for all World Cups across history, while yes is natural for all matches in the 2018 tournament. The question does not identify its time scope."
    },
    {
      "id": "boolq-dev-1662",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly says the injury can occur on lesser toes."
    },
    {
      "id": "boolq-dev-2868",
      "label": "no",
      "category": "time_or_scope",
      "evidence": "Passage explicitly says no permit is required for open handgun carry in North Carolina. This is passage-era legal text; no current-law verification was performed."
    },
    {
      "id": "boolq-dev-1447",
      "label": "yes",
      "category": "supported",
      "evidence": "A virtual image is described as formed yet unable to be projected onto a screen."
    },
    {
      "id": "boolq-dev-0383",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly gives the host an automatic berth. It repeats the passage in case 0659 with a paraphrased question."
    },
    {
      "id": "boolq-dev-2665",
      "label": "no",
      "category": "supported",
      "evidence": "The town name is explicitly described as fictional; the real place mentioned is a different bay/inlet."
    },
    {
      "id": "boolq-dev-2041",
      "label": "no",
      "category": "supported",
      "evidence": "The passage defines an unlisted public company as a public company not listed on any exchange."
    },
    {
      "id": "boolq-dev-1296",
      "label": "no",
      "category": "ambiguous",
      "evidence": "Passage describes the history of rainforest being called jungle and a broad definition of jungle; it does not directly settle equivalence. The no label is plausible under a strict distinction, but the passage also supports colloquial overlap."
    },
    {
      "id": "boolq-dev-1114",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly calls gray wolf also known as timber wolf."
    },
    {
      "id": "boolq-dev-1953",
      "label": "no",
      "category": "supported",
      "evidence": "Passage lists three main Australian time zones and different external-territory zones."
    },
    {
      "id": "boolq-dev-2493",
      "label": "yes",
      "category": "time_or_scope",
      "evidence": "Passage supports any-person arrest in England and Wales. That supports an existential reading of in the UK, but does not establish every UK jurisdiction or current eligibility conditions."
    },
    {
      "id": "boolq-dev-3009",
      "label": "yes",
      "category": "time_or_scope",
      "evidence": "Passage explicitly says service is offered 24 hours daily, while some routes are part-time. It is a source-era system-wide statement, not a guarantee for every route or current service."
    },
    {
      "id": "boolq-dev-2504",
      "label": "no",
      "category": "supported",
      "evidence": "Passage says Ian McKellen was offered the role and turned it down, while identifying him as Gandalf. The intended role is supplied by the question."
    },
    {
      "id": "boolq-dev-3125",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage identifies an optical phenomenon called an upside-down rainbow, while distinguishing it physically from rainbows."
    },
    {
      "id": "boolq-dev-1092",
      "label": "no",
      "category": "supported",
      "evidence": "Passage lists the six New England states and describes New York as a bordering state."
    },
    {
      "id": "boolq-dev-0055",
      "label": "no",
      "category": "ambiguous",
      "evidence": "Passage describes kissing and mutual feelings in season 3 but a final choice of Stefan. Get together could mean a romantic encounter or becoming a couple; the no label uses the latter sense."
    },
    {
      "id": "boolq-dev-2791",
      "label": "no",
      "category": "supported",
      "evidence": "Manhattan is listed as one of five constituent boroughs of New York City."
    },
    {
      "id": "boolq-dev-1842",
      "label": "yes",
      "category": "contradiction",
      "evidence": "Published label is yes, but the passage explicitly describes carbon as nonmetallic. Under the ordinary meaning of the question, passage-supported answer is no."
    },
    {
      "id": "boolq-dev-2951",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage gives explicit style guidance allowing actor for both women and men."
    },
    {
      "id": "boolq-dev-0852",
      "label": "no",
      "category": "ambiguous",
      "evidence": "Only passage: the Gold Award is often compared to Eagle Scout. This does not state eligibility rules, whether a person can join both organizations, or the relevant date. It cannot establish the no label by itself."
    },
    {
      "id": "boolq-dev-1171",
      "label": "no",
      "category": "time_or_scope",
      "evidence": "Passage treats Turkey and the EU as separate partners with a common border and customs union; the no label is consistent with that source-era reading. Present membership was not checked."
    },
    {
      "id": "boolq-dev-1550",
      "label": "no",
      "category": "supported",
      "evidence": "Passage explicitly says the relevant part of the Ring of Fire excludes Australia."
    },
    {
      "id": "boolq-dev-2116",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly states that Father's Day is celebrated in the Netherlands."
    },
    {
      "id": "boolq-dev-0341",
      "label": "no",
      "category": "supported",
      "evidence": "Passage explicitly says sound speed varies by substance and provides different values."
    },
    {
      "id": "boolq-dev-2098",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage includes siblings-in-law among relations connected by marriage."
    },
    {
      "id": "boolq-dev-2597",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage identifies Hellmann's and Best Foods as names for the same product line. The yes label follows the intended common-brand reading, rather than a separately verified legal-entity claim."
    },
    {
      "id": "boolq-dev-0468",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage describes floating plastic and debris in the North Pacific; mass is read as a collection, not a solid island."
    },
    {
      "id": "boolq-dev-0065",
      "label": "yes",
      "category": "supported",
      "evidence": "The angle-sum statement and acute/obtuse definitions support at least two acute angles for ordinary nondegenerate Euclidean triangles."
    },
    {
      "id": "boolq-dev-3114",
      "label": "yes",
      "category": "supported",
      "evidence": "The opening sentence explicitly calls the play a comedy."
    },
    {
      "id": "boolq-dev-0877",
      "label": "yes",
      "category": "ambiguous",
      "evidence": "Passage equates federal and national government within a federation. The question leaves country and constitutional structure unspecified, so universal equivalence is not established."
    },
    {
      "id": "boolq-dev-2630",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly lists garbanzo bean as another name for chickpea."
    },
    {
      "id": "boolq-dev-0448",
      "label": "yes",
      "category": "ambiguous",
      "evidence": "Passage describes a song about obsession, jealousy and surveillance, consistent with stalking. It never names the song, so the question's exact title/reference cannot be established from this passage alone."
    },
    {
      "id": "boolq-dev-3134",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage defines stateless people as not considered nationals by any state."
    },
    {
      "id": "boolq-dev-0525",
      "label": "no",
      "category": "supported",
      "evidence": "Passage identifies CAD as Canada's currency and distinguishes it from other dollar-denominated currencies."
    },
    {
      "id": "boolq-dev-1381",
      "label": "no",
      "category": "time_or_scope",
      "evidence": "Passage prohibits plate blocking without the ball in MLB from the 2014 season. The question does not specify league or date; other rulesets are not established."
    },
    {
      "id": "boolq-dev-0308",
      "label": "no",
      "category": "supported",
      "evidence": "Passage describes coriander being used together with cumin, treating them as separate ingredients."
    },
    {
      "id": "boolq-dev-2849",
      "label": "no",
      "category": "ambiguous",
      "evidence": "Passage supports that the stinger is not always left lodged in skin. If stinger in a bee sting means the apparatus that delivers the sting, the question asks something different. The intended retained-stinger meaning is unstated."
    },
    {
      "id": "boolq-dev-0210",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage describes pregnancy, choosing to have the child, and the baby shower; supports the intended storyline answer."
    },
    {
      "id": "boolq-dev-3115",
      "label": "yes",
      "category": "ambiguous",
      "evidence": "Passage permits multiple citizenship under Czech law from 2014 but does not state US rules or the questioner's eligibility. Yes is plausible for possibility, but a personal can-I conclusion is not established by the passage."
    },
    {
      "id": "boolq-dev-0587",
      "label": "yes",
      "category": "ambiguous",
      "evidence": "Passage says Boost and Virgin were two brands reorganized into one Sprint group in 2010. Same parent versus same brand/service gives different readings; the ownership frame is also dated."
    },
    {
      "id": "boolq-dev-2552",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly calls lily-of-the-valley highly poisonous."
    },
    {
      "id": "boolq-dev-1859",
      "label": "no",
      "category": "supported",
      "evidence": "Passage describes Pinot gris as a distinct variety thought to be a mutant clone of Pinot noir."
    },
    {
      "id": "boolq-dev-1739",
      "label": "yes",
      "category": "ambiguous",
      "evidence": "Passage places spongy bone in the walls of the medullary cavity and marrow in the space itself. In the cavity can include its walls or mean only its contents; yes adopts the former."
    },
    {
      "id": "boolq-dev-1066",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage describes the preamble as the Constitution's introductory statement, adopted by the Constituent Assembly with it."
    },
    {
      "id": "boolq-dev-3247",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly says the fifth season premiered in 2018; this establishes existence without relying on its future-release details."
    },
    {
      "id": "boolq-dev-1335",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly includes optic and olfactory nerves as parts of the CNS under a stated common classification."
    },
    {
      "id": "boolq-dev-1979",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage defines insulin resistance as a pathological condition."
    },
    {
      "id": "boolq-dev-2426",
      "label": "yes",
      "category": "time_or_scope",
      "evidence": "Passage explicitly states no duty to answer police questions outside detention or arrest, in a US/Fifth-Amendment context. The question does not name a jurisdiction; no current legal verification was performed."
    },
    {
      "id": "boolq-dev-3165",
      "label": "yes",
      "category": "ambiguous",
      "evidence": "Passage only describes a 2010 US bottling-size reduction. It names no product and does not distinguish Red Stripe Light from another beverage or current availability. The yes label is unsupported by this excerpt alone."
    },
    {
      "id": "boolq-dev-0980",
      "label": "no",
      "category": "supported",
      "evidence": "Passage identifies pyruvate as the conjugate base of pyruvic acid, distinguishing the chemical forms."
    },
    {
      "id": "boolq-dev-2011",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage explicitly says Howard joined Houston and spent three seasons with the Rockets."
    },
    {
      "id": "boolq-dev-2818",
      "label": "yes",
      "category": "supported",
      "evidence": "Passage reports yield increases under rotation; uncertain causal explanation does not negate the stated increase."
    },
    {
      "id": "boolq-dev-1603",
      "label": "no",
      "category": "supported",
      "evidence": "Passage explicitly describes Father's Day on different dates across countries."
    },
    {
      "id": "boolq-dev-1625",
      "label": "yes",
      "category": "ambiguous",
      "evidence": "Passage says black garlic softness increases with water content. It does not establish a normal or desired mushy texture, or a specifically Japanese preparation."
    },
    {
      "id": "boolq-dev-0767",
      "label": "yes",
      "category": "ambiguous",
      "evidence": "Passage compares playing-field widths, not whole-stadium size or seating capacity. Bigger stadium is broader than the evidence; yes assumes a field-size interpretation."
    },
    {
      "id": "boolq-dev-1570",
      "label": "no",
      "category": "ambiguous",
      "evidence": "Passage establishes a touchscreen phone in 1994 but gives no iPhone release date. The no answer requires background knowledge not contained in the supplied excerpt."
    }
  ],
  "main": {
    "schema_version": 1,
    "generated_at_utc": "2026-10-02T08:47:47.979378+00:00",
    "metadata": {
      "models": {
        "Jev": "typesafe/jev-1.13",
        "Luna": "openai/gpt-6-luna"
      },
      "created_at": "2026-10-02T08:42:23.244948+00:00",
      "seed": 20261002,
      "cases": 180,
      "calls_planned": 360,
      "budget_usd": 0.999912658,
      "luna_reasoning": "none",
      "luna_provider": "OpenAI",
      "protocol": "One non-streaming request per model per case; sequential paired random order; one shared persistent HTTPS connection; no automatic retries; two warmups per model excluded from measured results.",
      "cases_sha256": "60e6db60c72869936548c3c4620aa3b070cf12e516f3be42a61a306d0f592660",
      "runner_sha256": "3a05cb20816fbc5b7c30a065429a7365afa9583f273d683bf69784f4af3e9ad7",
      "status": "complete",
      "charged_usd": 0.011550069999999994,
      "calls_completed": 364,
      "finished_at": "2026-10-02T08:46:47.575786+00:00"
    },
    "totals": {
      "measured_requests": 360,
      "measured_valid": 360,
      "warmup_requests": 4,
      "all_cost": {
        "known_total_usd": 0.01155007,
        "known_count": 364,
        "missing_count": 0,
        "complete": true,
        "total_usd": 0.01155007,
        "per_1000_requests_usd": 0.03173096153846154
      },
      "measured_cost": {
        "known_total_usd": 0.011462728,
        "known_count": 360,
        "missing_count": 0,
        "complete": true,
        "total_usd": 0.011462728,
        "per_1000_requests_usd": 0.03184091111111111
      },
      "warmup_cost": {
        "known_total_usd": 8.7342e-05,
        "known_count": 4,
        "missing_count": 0,
        "complete": true,
        "total_usd": 8.7342e-05,
        "per_1000_requests_usd": 0.0218355
      }
    },
    "experiments": {
      "boolq": {
        "target_kind": "label",
        "models": {
          "Jev": {
            "requests": 60,
            "unique_cases": 60,
            "valid_count": 60,
            "invalid_count": 0,
            "success_rate": 1.0,
            "target_kind": "label",
            "quality": {
              "accuracy": 0.95,
              "brier": 0.06835999999999999,
              "log_loss": 0.2482454306996313,
              "correct_count": 57,
              "accuracy_all_requests": 0.95
            },
            "valid_response_latency_s": {
              "count": 60,
              "p50": 0.23620002050301991,
              "p95": 0.37197747530590275,
              "mean": 0.26132939653180076
            },
            "all_request_latency_s": {
              "count": 60,
              "p50": 0.23620002050301991,
              "p95": 0.37197747530590275,
              "mean": 0.26132939653180076
            },
            "cost": {
              "known_total_usd": 0.001190826,
              "known_count": 60,
              "missing_count": 0,
              "complete": true,
              "total_usd": 0.001190826,
              "per_1000_requests_usd": 0.019847100000000003,
              "per_correct_answer_usd": 2.0891684210526317e-05
            },
            "response_models": [
              "typesafe/jev-1.13-20260917"
            ],
            "providers": [
              "TypeSafe"
            ]
          },
          "Luna": {
            "requests": 60,
            "unique_cases": 60,
            "valid_count": 60,
            "invalid_count": 0,
            "success_rate": 1.0,
            "target_kind": "label",
            "quality": {
              "accuracy": 0.8166666666666667,
              "brier": 0.16761500000000001,
              "log_loss": 3.6759119683872865,
              "correct_count": 49,
              "accuracy_all_requests": 0.8166666666666667
            },
            "valid_response_latency_s": {
              "count": 60,
              "p50": 1.1484179999970365,
              "p95": 1.4701433298032498,
              "mean": 1.1566619764028776
            },
            "all_request_latency_s": {
              "count": 60,
              "p50": 1.1484179999970365,
              "p95": 1.4701433298032498,
              "mean": 1.1566619764028776
            },
            "cost": {
              "known_total_usd": 0.002669,
              "known_count": 60,
              "missing_count": 0,
              "complete": true,
              "total_usd": 0.002669,
              "per_1000_requests_usd": 0.04448333333333333,
              "per_correct_answer_usd": 5.446938775510204e-05
            },
            "response_models": [
              "openai/gpt-6-luna"
            ],
            "providers": [
              "OpenAI"
            ]
          }
        },
        "paired": {
          "difference_direction": "Luna minus Jev",
          "latency_ratio_direction": "Luna divided by Jev",
          "unit": "case; valid repeats averaged within case, latency repeats reduced to median",
          "total_cases": 60,
          "paired_cases": 60,
          "excluded_cases": 0,
          "mismatched_target_case_ids": [],
          "bootstrap_samples": 2000,
          "bootstrap_seed": 3745107185,
          "confidence_level": 0.95,
          "quality_differences": {
            "accuracy": {
              "estimate": -0.13333333333333333,
              "ci95": [
                -0.21666666666666667,
                -0.05
              ],
              "favours_luna_when": "positive"
            },
            "brier": {
              "estimate": 0.09925500000000001,
              "ci95": [
                0.03473529166666667,
                0.17184033333333323
              ],
              "favours_luna_when": "negative"
            },
            "log_loss": {
              "estimate": 3.4276665376876547,
              "ci95": [
                1.14736590607063,
                6.148670921720271
              ],
              "favours_luna_when": "negative"
            }
          },
          "latency_median_ratio": {
            "estimate": 4.862057156266646,
            "ci95": [
              4.41750089230697,
              5.1040566772845715
            ],
            "paired_cases": 60,
            "defined_bootstrap_samples": 2000,
            "favours_luna_when": "less than 1"
          }
        }
      },
      "policy": {
        "target_kind": "label",
        "models": {
          "Jev": {
            "requests": 60,
            "unique_cases": 60,
            "valid_count": 60,
            "invalid_count": 0,
            "success_rate": 1.0,
            "target_kind": "label",
            "quality": {
              "accuracy": 1.0,
              "brier": 0.008311666666666669,
              "log_loss": 0.06748769727937004,
              "correct_count": 60,
              "accuracy_all_requests": 1.0
            },
            "valid_response_latency_s": {
              "count": 60,
              "p50": 0.24326154148729984,
              "p95": 0.3312737014319282,
              "mean": 0.25546772780217
            },
            "all_request_latency_s": {
              "count": 60,
              "p50": 0.24326154148729984,
              "p95": 0.3312737014319282,
              "mean": 0.25546772780217
            },
            "cost": {
              "known_total_usd": 0.001255632,
              "known_count": 60,
              "missing_count": 0,
              "complete": true,
              "total_usd": 0.001255632,
              "per_1000_requests_usd": 0.020927200000000003,
              "per_correct_answer_usd": 2.0927200000000002e-05
            },
            "response_models": [
              "typesafe/jev-1.13-20260917"
            ],
            "providers": [
              "TypeSafe"
            ]
          },
          "Luna": {
            "requests": 60,
            "unique_cases": 60,
            "valid_count": 60,
            "invalid_count": 0,
            "success_rate": 1.0,
            "target_kind": "label",
            "quality": {
              "accuracy": 1.0,
              "brier": 0.0,
              "log_loss": 9.99600361081321e-16,
              "correct_count": 60,
              "accuracy_all_requests": 1.0
            },
            "valid_response_latency_s": {
              "count": 60,
              "p50": 1.1491409375012154,
              "p95": 1.6466312964330425,
              "mean": 1.1813501993514364
            },
            "all_request_latency_s": {
              "count": 60,
              "p50": 1.1491409375012154,
              "p95": 1.6466312964330425,
              "mean": 1.1813501993514364
            },
            "cost": {
              "known_total_usd": 0.0027441,
              "known_count": 60,
              "missing_count": 0,
              "complete": true,
              "total_usd": 0.0027441,
              "per_1000_requests_usd": 0.045735000000000005,
              "per_correct_answer_usd": 4.5735e-05
            },
            "response_models": [
              "openai/gpt-6-luna"
            ],
            "providers": [
              "OpenAI"
            ]
          }
        },
        "paired": {
          "difference_direction": "Luna minus Jev",
          "latency_ratio_direction": "Luna divided by Jev",
          "unit": "case; valid repeats averaged within case, latency repeats reduced to median",
          "total_cases": 60,
          "paired_cases": 60,
          "excluded_cases": 0,
          "mismatched_target_case_ids": [],
          "bootstrap_samples": 2000,
          "bootstrap_seed": 2184452498,
          "confidence_level": 0.95,
          "quality_differences": {
            "accuracy": {
              "estimate": 0.0,
              "ci95": [
                0.0,
                0.0
              ],
              "favours_luna_when": "positive"
            },
            "brier": {
              "estimate": -0.008311666666666669,
              "ci95": [
                -0.015043791666666669,
                -0.0037815833333333347
              ],
              "favours_luna_when": "negative"
            },
            "log_loss": {
              "estimate": -0.06748769727936903,
              "ci95": [
                -0.09047705235375375,
                -0.04933797508124458
              ],
              "favours_luna_when": "negative"
            }
          },
          "latency_median_ratio": {
            "estimate": 4.723890716450177,
            "ci95": [
              4.262462643633293,
              5.045807495730796
            ],
            "paired_cases": 60,
            "defined_bootstrap_samples": 2000,
            "favours_luna_when": "less than 1"
          }
        }
      },
      "probability": {
        "target_kind": "probability",
        "models": {
          "Jev": {
            "requests": 60,
            "unique_cases": 60,
            "valid_count": 60,
            "invalid_count": 0,
            "success_rate": 1.0,
            "target_kind": "probability",
            "quality": {
              "mae": 0.0644286219101022,
              "excess_brier": 0.009976842577220275,
              "expected_brier": 0.15633600538274284,
              "rmse": 0.09988414577509423
            },
            "valid_response_latency_s": {
              "count": 60,
              "p50": 0.237314312485978,
              "p95": 0.31138571220799344,
              "mean": 0.2471950327322702
            },
            "all_request_latency_s": {
              "count": 60,
              "p50": 0.237314312485978,
              "p95": 0.31138571220799344,
              "mean": 0.2471950327322702
            },
            "cost": {
              "known_total_usd": 0.00111867,
              "known_count": 60,
              "missing_count": 0,
              "complete": true,
              "total_usd": 0.00111867,
              "per_1000_requests_usd": 0.0186445,
              "per_correct_answer_usd": null
            },
            "response_models": [
              "typesafe/jev-1.13-20260917"
            ],
            "providers": [
              "TypeSafe"
            ]
          },
          "Luna": {
            "requests": 60,
            "unique_cases": 60,
            "valid_count": 60,
            "invalid_count": 0,
            "success_rate": 1.0,
            "target_kind": "probability",
            "quality": {
              "mae": 0.03277039096880469,
              "excess_brier": 0.005510213165512707,
              "expected_brier": 0.15186937597103528,
              "rmse": 0.07423081008255741
            },
            "valid_response_latency_s": {
              "count": 60,
              "p50": 1.1426221465080744,
              "p95": 1.5959135121971477,
              "mean": 1.2395371187997322
            },
            "all_request_latency_s": {
              "count": 60,
              "p50": 1.1426221465080744,
              "p95": 1.5959135121971477,
              "mean": 1.2395371187997322
            },
            "cost": {
              "known_total_usd": 0.0024845,
              "known_count": 60,
              "missing_count": 0,
              "complete": true,
              "total_usd": 0.0024845,
              "per_1000_requests_usd": 0.04140833333333333,
              "per_correct_answer_usd": null
            },
            "response_models": [
              "openai/gpt-6-luna"
            ],
            "providers": [
              "OpenAI"
            ]
          }
        },
        "paired": {
          "difference_direction": "Luna minus Jev",
          "latency_ratio_direction": "Luna divided by Jev",
          "unit": "case; valid repeats averaged within case, latency repeats reduced to median",
          "total_cases": 60,
          "paired_cases": 60,
          "excluded_cases": 0,
          "mismatched_target_case_ids": [],
          "bootstrap_samples": 2000,
          "bootstrap_seed": 961937342,
          "confidence_level": 0.95,
          "quality_differences": {
            "mae": {
              "estimate": -0.0316582309412975,
              "ci95": [
                -0.0521207929380925,
                -0.01113896868529326
              ],
              "favours_luna_when": "negative"
            },
            "excess_brier": {
              "estimate": -0.004466629411707567,
              "ci95": [
                -0.011295080296257799,
                0.002042924425773913
              ],
              "favours_luna_when": "negative"
            },
            "expected_brier": {
              "estimate": -0.004466629411707568,
              "ci95": [
                -0.011295080296257797,
                0.0020429244257739133
              ],
              "favours_luna_when": "negative"
            }
          },
          "latency_median_ratio": {
            "estimate": 4.814805034464946,
            "ci95": [
              4.469474257664341,
              5.36853956771415
            ],
            "paired_cases": 60,
            "defined_bootstrap_samples": 2000,
            "favours_luna_when": "less than 1"
          }
        }
      }
    },
    "definitions": {
      "success": "valid=true with a finite probability and valid target in [0,1]; label targets must be 0 or 1",
      "accuracy": "Fraction correct among valid responses, predicting yes when p_yes >= 0.5",
      "accuracy_all_requests": "Correct valid responses divided by every measured request; invalid requests count as failures",
      "brier": "Mean (p_yes - observed_label)^2; lower is better",
      "log_loss": "Mean binary negative log likelihood in natural-log units; p clipped to [1e-15, 1-1e-15]",
      "mae": "Mean absolute difference between reported p_yes and exact reference probability",
      "rmse": "Square root of mean squared error against exact reference probabilities",
      "excess_brier": "Mean (p_yes - true_probability)^2: expected Brier above the optimal probability forecast",
      "expected_brier": "Mean [(p_yes - true_probability)^2 + true_probability*(1-true_probability)]",
      "latency": "End-to-end request time through receipt of a complete response; main table includes valid responses only",
      "cost": "Recorded billed USD across all requests in the stated phase, including failed/invalid requests when billed",
      "cost_per_1000": "Measured total billed cost / measured request count * 1000; withheld if any bill is missing",
      "cost_per_correct": "All measured billed cost / number of correct valid label responses; withheld for missing bills or zero correct",
      "bootstrap": "Percentile 95% paired bootstrap by case identity; quality repeats averaged within case; latency repeats use median; no CI for fewer than 2 pairs"
    },
    "caveats": [
      "This is a small benchmark. Intervals describe variation across the sampled cases; they do not prove a general model ranking.",
      "Luna supplies generated numerical probabilities; Jev supplies its native Noul probabilities. Neither adapter extracts next-token probabilities. Brier combines calibration and discrimination; this small sample cannot establish general calibration.",
      "Each task has a separate quality scale. Quality is not pooled across tasks; no overall winner is inferred.",
      "BoolQ is public and may have appeared in training data. Reading-comprehension results can be affected by contamination.",
      "Paired comparisons include only cases with valid responses from both models. Check success rates and excluded pairs before comparing conditional quality.",
      "Latency includes the OpenRouter/provider/network path. Provider routing, caching, load, reasoning settings, and output length can change speed and price.",
      "The intended primary protocol is Luna reasoning=none, sequential randomized interleaving, and no retries. The recorded run metadata is the source of truth for whether that protocol was used.",
      "95% intervals are exploratory percentile bootstrap intervals without multiple-comparison correction. Zero-width intervals can occur in a small or uniform sample.",
      "Costs are USD as reported by the runner. Missing billing remains unknown; cost per 1,000 is a sample-based projection, not a price guarantee."
    ]
  },
  "cache": {
    "metadata": {
      "experiment": "cache_policy",
      "created_at": "2026-10-02T08:49:58.826529+00:00",
      "models": {
        "Luna cached": "openai/gpt-6-luna",
        "Luna uncached": "openai/gpt-6-luna",
        "Jev": "typesafe/jev-1.13"
      },
      "seed": 20261002,
      "cases": 24,
      "calls_planned": 73,
      "budget_usd": 0.988362588,
      "charged_usd": 0.012438148999999999,
      "calls_completed": 73,
      "status": "complete",
      "rulebook_sha256": "01006795edc5f9ee13a1e4efc56fc8e19516d74f40cddbce2e72d19b00f0d43b",
      "cases_sha256": "7559a18a2c58c15206a951c2a483cae6eff11969a62b4ab799a87e919ab2c8ea",
      "code_sha256": "40976904f39e343bfcf97ad38815f70cfe9e10cd69ef2260d77bad57060d14f7",
      "rulebook_words": 1903,
      "prompt_cache_key": "luna-jev-policy-1e70847917e24fd092c9b043bb8de051",
      "source": "https://openrouter.ai/docs/guides/best-practices/prompt-caching",
      "luna_reasoning": "none",
      "luna_provider": "OpenAI",
      "protocol": "One distinct Luna cached priming call, then 24 sequential cases with all three arms randomly ordered within case; no automatic retries; actual usage cost and cache counters; exact same Luna message text and segmentation.",
      "finished_at": "2026-10-02T08:50:56.490676+00:00"
    },
    "arms": {
      "Luna cached": {
        "n": 24,
        "valid": 24,
        "cost_observed_n": 24,
        "accuracy_including_invalid_as_wrong": 0.875,
        "brier_valid_only": 0.125,
        "median_latency_s": 0.9179664584953571,
        "p95_latency_s": 1.1803516982632571,
        "known_cost_subtotal_usd": 0.00171132,
        "cost_missing_n": 0,
        "total_cost_usd": 0.00171132,
        "mean_cost_usd": 7.1305e-05,
        "cache_observed_n": 24,
        "cache_hit_calls": 24,
        "cache_read_tokens": 60792,
        "known_cache_read_subtotal_tokens": 60792,
        "cache_write_observed_n": 24,
        "cache_write_tokens": 0,
        "known_cache_write_subtotal_tokens": 0,
        "prompt_tokens": 69426,
        "cache_read_fraction": 0.8756373692852821
      },
      "Luna uncached": {
        "n": 24,
        "valid": 24,
        "cost_observed_n": 24,
        "accuracy_including_invalid_as_wrong": 0.875,
        "brier_valid_only": 0.125,
        "median_latency_s": 1.0602054579940159,
        "p95_latency_s": 1.291354949955712,
        "known_cost_subtotal_usd": 0.0071802,
        "cost_missing_n": 0,
        "total_cost_usd": 0.0071802,
        "mean_cost_usd": 0.000299175,
        "cache_observed_n": 24,
        "cache_hit_calls": 0,
        "cache_read_tokens": 0,
        "known_cache_read_subtotal_tokens": 0,
        "cache_write_observed_n": 24,
        "cache_write_tokens": 0,
        "known_cache_write_subtotal_tokens": 0,
        "prompt_tokens": 69402,
        "cache_read_fraction": 0.0
      },
      "Jev": {
        "n": 24,
        "valid": 24,
        "cost_observed_n": 24,
        "accuracy_including_invalid_as_wrong": 0.8333333333333334,
        "brier_valid_only": 0.08786666666666666,
        "median_latency_s": 0.2544942910026293,
        "p95_latency_s": 0.33727485434792465,
        "known_cost_subtotal_usd": 0.003184104,
        "cost_missing_n": 0,
        "total_cost_usd": 0.003184104,
        "mean_cost_usd": 0.000132671,
        "cache_observed_n": 0,
        "cache_hit_calls": 0,
        "cache_read_tokens": null,
        "known_cache_read_subtotal_tokens": 0,
        "cache_write_observed_n": 0,
        "cache_write_tokens": null,
        "known_cache_write_subtotal_tokens": 0,
        "prompt_tokens": null,
        "cache_read_fraction": null
      }
    },
    "paired": {
      "Luna cached / Luna uncached": {
        "n": 24,
        "median_latency_ratio": 0.9125562576625701,
        "total_cost_ratio": 0.2383387649369098,
        "label_agreement": 0.9166666666666666
      },
      "Luna cached / Jev": {
        "n": 24,
        "median_latency_ratio": 3.617968437782495,
        "total_cost_ratio": 0.5374573192332913,
        "label_agreement": 0.875
      },
      "Luna uncached / Jev": {
        "n": 24,
        "median_latency_ratio": 3.918573737604335,
        "total_cost_ratio": 2.255014283453053,
        "label_agreement": 0.875
      }
    },
    "cache_verification": {
      "cached_arm_has_observed_hits": true,
      "uncached_control_verified": true,
      "controlled_cache_comparison_verified": true,
      "status": "observed cache hits and clean uncached control",
      "prime_cached_tokens": 0,
      "prime_cache_write_tokens": 2533,
      "prime_cost_usd": 0.000362525,
      "prime_latency_s": 1.2044517089962028,
      "prime_was_observed_cold": true,
      "setup_inclusive_cached_cost_usd": 0.002073845,
      "setup_inclusive_cost_per_measured_case_usd": 8.641020833333333e-05
    }
  },
  "manifest": {
    "boolq": {
      "attribution": "BoolQ: Exploring the Surprising Difficulty of Natural Yes/No Questions. Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, Kristina Toutanova. NAACL 2019.",
      "license": "CC BY-SA 3.0",
      "license_url": "https://creativecommons.org/licenses/by-sa/3.0/",
      "mirror": "https://huggingface.co/datasets/google/boolq",
      "mirror_rows_api": "https://datasets-server.huggingface.co/rows?dataset=google%2Fboolq&config=default&split=validation",
      "official_download": "https://storage.googleapis.com/boolq/dev.jsonl",
      "paper": "https://arxiv.org/abs/1905.10044",
      "raw_jsonl_sha256": "e8fb84fbf510b022e963cddf3a3aded04151afa0ea0ef1cc1bf22f260ddd2344",
      "retrieval_note": "Original Google Cloud Storage URL returned HTTP 403. Downloaded complete Google-owned Hugging Face validation mirror via rows API, then serialized to JSONL. This JSONL hash is of the local normalized mirror, not the original GCS file.",
      "sample_method": "random.Random(seed).sample(range(3270), 60); uniform without replacement; no filtering by label or model output",
      "selected_row_indices_zero_based": [
        659,
        2774,
        579,
        2217,
        1517,
        1662,
        2868,
        1447,
        383,
        2665,
        2041,
        1296,
        1114,
        1953,
        2493,
        3009,
        2504,
        3125,
        1092,
        55,
        2791,
        1842,
        2951,
        852,
        1171,
        1550,
        2116,
        341,
        2098,
        2597,
        468,
        65,
        3114,
        877,
        2630,
        448,
        3134,
        525,
        1381,
        308,
        2849,
        210,
        3115,
        587,
        2552,
        1859,
        1739,
        1066,
        3247,
        1335,
        1979,
        2426,
        3165,
        980,
        2011,
        2818,
        1603,
        1625,
        767,
        1570
      ],
      "source": "https://github.com/google-research-datasets/boolean-questions",
      "source_rows": 3270
    },
    "cases_file": "cases.jsonl",
    "cases_sha256": "60e6db60c72869936548c3c4620aa3b070cf12e516f3be42a61a306d0f592660",
    "count": 180,
    "distribution": {
      "boolq": {
        "count": 60,
        "labels": {
          "no": 24,
          "yes": 36
        },
        "target_max": 1.0,
        "target_mean": 0.6,
        "target_min": 0.0
      },
      "policy": {
        "count": 60,
        "labels": {
          "no": 30,
          "yes": 30
        },
        "target_max": 1.0,
        "target_mean": 0.5,
        "target_min": 0.0
      },
      "probability": {
        "count": 60,
        "target_max": 1.0,
        "target_mean": 0.4896041483485319,
        "target_min": 0.0
      }
    },
    "generator_version": 1,
    "limitations": [
      "This is a small, exploratory pilot, not a broad ranking of either model.",
      "BoolQ development labels are human annotations rather than a mathematical oracle; the public set may be present in model training data.",
      "Synthetic policy and probability cases measure the specified rules and probability problems, not general operational or forecasting performance.",
      "Probability cases use exact known event probabilities, not observed binary outcomes. Use MAE/RMSE and expected proper scores for these cases.",
      "Thresholds and prompts must be frozen before reading scored model outputs."
    ],
    "model_input_fields": [
      "state",
      "question",
      "criteria"
    ],
    "name": "Luna versus Jev matched binary decisions pilot",
    "never_send_to_models": [
      "id",
      "experiment",
      "target",
      "target_kind",
      "source"
    ],
    "policy_families": {
      "access": 20,
      "delivery": 20,
      "refund": 20
    },
    "probability_families": {
      "conditional_table": 20,
      "weighted_mixture": 20,
      "without_replacement": 20
    },
    "seed": 20261002
  }
}