{"case_id": "warmup-yes", "experiment": "warmup", "target": 1, "target_kind": "label", "model_label": "Jev", "requested_model": "typesafe/jev-1.13", "phase": "warmup", "sequence": 0, "started_at": "2026-10-02T08:40:29.938428+00:00", "request": {"model": "typesafe/jev-1.13", "state": {"rule": "Approve if the number is at least 10.", "number": 15}, "questions": {"answer": {"type": "noul", "instructions": "Is approval required under the rule?", "criteria": {"true": "The rule requires approval.", "false": "The rule does not require approval."}}}}, "endpoint": "/api/alpha/decisions", "probability": 0.96, "valid": true, "error": null, "cost_usd": 1.3692e-05, "usage": {"input_tokens": 326, "output_tokens": 20, "cost": 1.3692e-05}, "response_model": "typesafe/jev-1.13-20260917", "provider": "TypeSafe", "http_status": 200, "response": {"model": "typesafe/jev-1.13-20260917", "answers": {"answer": {"type": "noul", "noul": 0.96}}, "usage": {"input_tokens": 326, "output_tokens": 20, "cost": 1.3692e-05}, "id": "gen-dec-1790930430-PnXUyR10yToLic6TzRUL", "provider": "TypeSafe"}, "generation_id": "gen-dec-1790930430-PnXUyR10yToLic6TzRUL", "answer": {"type": "noul", "noul": 0.96}, "latency_s": 0.5134258330217563}
{"case_id": "warmup-yes", "experiment": "warmup", "target": 1, "target_kind": "label", "model_label": "Luna", "requested_model": "openai/gpt-6-luna", "phase": "warmup", "sequence": 1, "started_at": "2026-10-02T08:40:30.451938+00:00", "request": {"model": "openai/gpt-6-luna", "messages": [{"role": "system", "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."}, {"role": "user", "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The rule does not require approval.\", \"true\": \"The rule requires approval.\"}, \"instructions\": \"Is approval required under the rule?\", \"type\": \"noul\"}}, \"state\": {\"number\": 15, \"rule\": \"Approve if the number is at least 10.\"}}"}], "response_format": {"type": "json_schema", "json_schema": {"name": "noul_answer", "strict": true, "schema": {"type": "object", "properties": {"type": {"type": "string", "enum": ["noul"]}, "noul": {"type": "number", "minimum": 0, "maximum": 1}}, "required": ["type", "noul"], "additionalProperties": false}}}, "reasoning": {"effort": "none"}, "max_tokens": 128, "prompt_cache_options": {"mode": "explicit"}, "provider": {"only": ["OpenAI"], "allow_fallbacks": false, "require_parameters": true}, "stream": false}, "endpoint": "/api/v1/chat/completions", "probability": 1.0, "valid": true, "error": null, "cost_usd": 3e-05, "usage": {"prompt_tokens": 200, "completion_tokens": 20, "total_tokens": 220, "cost": 3e-05, "is_byok": false, "prompt_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0, "audio_tokens": 0, "video_tokens": 0}, "cost_details": {"upstream_inference_cost": 3e-05, "upstream_inference_prompt_cost": 2e-05, "upstream_inference_completions_cost": 1e-05}, "completion_tokens_details": {"reasoning_tokens": 0, "image_tokens": 0, "audio_tokens": 0}}, "response_model": "openai/gpt-6-luna", "provider": "OpenAI", "http_status": 200, "response": {"id": "gen-1790930430-6KHg49gHWoLE4uOHARxX", "object": "chat.completion", "created": 1790930430, "model": "openai/gpt-6-luna", "provider": "OpenAI", "system_fingerprint": null, "service_tier": "default", "choices": [{"index": 0, "logprobs": null, "finish_reason": "stop", "native_finish_reason": "completed", "message": {"role": "assistant", "content": "{\"type\":\"noul\",\"noul\":1}", "refusal": null, "reasoning": null}}], "usage": {"prompt_tokens": 200, "completion_tokens": 20, "total_tokens": 220, "cost": 3e-05, "is_byok": false, "prompt_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0, "audio_tokens": 0, "video_tokens": 0}, "cost_details": {"upstream_inference_cost": 3e-05, "upstream_inference_prompt_cost": 2e-05, "upstream_inference_completions_cost": 1e-05}, "completion_tokens_details": {"reasoning_tokens": 0, "image_tokens": 0, "audio_tokens": 0}}}, "generation_id": "gen-1790930430-6KHg49gHWoLE4uOHARxX", "finish_reason": "stop", "answer": {"type": "noul", "noul": 1}, "latency_s": 1.2567277500056662}
{"case_id": "warmup-no", "experiment": "warmup", "target": 0, "target_kind": "label", "model_label": "Jev", "requested_model": "typesafe/jev-1.13", "phase": "warmup", "sequence": 2, "started_at": "2026-10-02T08:40:31.708754+00:00", "request": {"model": "typesafe/jev-1.13", "state": {"rule": "Approve if the number is at least 10.", "number": 4}, "questions": {"answer": {"type": "noul", "instructions": "Is approval required under the rule?", "criteria": {"true": "The rule requires approval.", "false": "The rule does not require approval."}}}}, "endpoint": "/api/alpha/decisions", "probability": 0.04, "valid": true, "error": null, "cost_usd": 1.365e-05, "usage": {"input_tokens": 325, "output_tokens": 20, "cost": 1.365e-05}, "response_model": "typesafe/jev-1.13-20260917", "provider": "TypeSafe", "http_status": 200, "response": {"model": "typesafe/jev-1.13-20260917", "answers": {"answer": {"type": "noul", "noul": 0.04}}, "usage": {"input_tokens": 325, "output_tokens": 20, "cost": 1.365e-05}, "id": "gen-dec-1790930431-4Pw2DZ3Li7d5cRIejyOA", "provider": "TypeSafe"}, "generation_id": "gen-dec-1790930431-4Pw2DZ3Li7d5cRIejyOA", "answer": {"type": "noul", "noul": 0.04}, "latency_s": 0.22932208300335333}
{"case_id": "warmup-no", "experiment": "warmup", "target": 0, "target_kind": "label", "model_label": "Luna", "requested_model": "openai/gpt-6-luna", "phase": "warmup", "sequence": 3, "started_at": "2026-10-02T08:40:31.938167+00:00", "request": {"model": "openai/gpt-6-luna", "messages": [{"role": "system", "content": "Answer the single typed question using the supplied state and criteria. Return only the answer object with exactly two fields: type (the string noul) and noul (a number from 0 to 1 representing the probability that the answer is yes). The probability of no is 1 minus noul. For a random experiment, estimate the event probability under the stated sampling process. Do not return explanations."}, {"role": "user", "content": "{\"questions\": {\"answer\": {\"criteria\": {\"false\": \"The rule does not require approval.\", \"true\": \"The rule requires approval.\"}, \"instructions\": \"Is approval required under the rule?\", \"type\": \"noul\"}}, \"state\": {\"number\": 4, \"rule\": \"Approve if the number is at least 10.\"}}"}], "response_format": {"type": "json_schema", "json_schema": {"name": "noul_answer", "strict": true, "schema": {"type": "object", "properties": {"type": {"type": "string", "enum": ["noul"]}, "noul": {"type": "number", "minimum": 0, "maximum": 1}}, "required": ["type", "noul"], "additionalProperties": false}}}, "reasoning": {"effort": "none"}, "max_tokens": 128, "prompt_cache_options": {"mode": "explicit"}, "provider": {"only": ["OpenAI"], "allow_fallbacks": false, "require_parameters": true}, "stream": false}, "endpoint": "/api/v1/chat/completions", "probability": 0.0, "valid": true, "error": null, "cost_usd": 3e-05, "usage": {"prompt_tokens": 200, "completion_tokens": 20, "total_tokens": 220, "cost": 3e-05, "is_byok": false, "prompt_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0, "audio_tokens": 0, "video_tokens": 0}, "cost_details": {"upstream_inference_cost": 3e-05, "upstream_inference_prompt_cost": 2e-05, "upstream_inference_completions_cost": 1e-05}, "completion_tokens_details": {"reasoning_tokens": 0, "image_tokens": 0, "audio_tokens": 0}}, "response_model": "openai/gpt-6-luna", "provider": "OpenAI", "http_status": 200, "response": {"id": "gen-1790930432-iz5x6x37CsXyiDtzpFs9", "object": "chat.completion", "created": 1790930432, "model": "openai/gpt-6-luna", "provider": "OpenAI", "system_fingerprint": null, "service_tier": "default", "choices": [{"index": 0, "logprobs": null, "finish_reason": "stop", "native_finish_reason": "completed", "message": {"role": "assistant", "content": "{\"type\":\"noul\",\"noul\":0}", "refusal": null, "reasoning": null}}], "usage": {"prompt_tokens": 200, "completion_tokens": 20, "total_tokens": 220, "cost": 3e-05, "is_byok": false, "prompt_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0, "audio_tokens": 0, "video_tokens": 0}, "cost_details": {"upstream_inference_cost": 3e-05, "upstream_inference_prompt_cost": 2e-05, "upstream_inference_completions_cost": 1e-05}, "completion_tokens_details": {"reasoning_tokens": 0, "image_tokens": 0, "audio_tokens": 0}}}, "generation_id": "gen-1790930432-iz5x6x37CsXyiDtzpFs9", "finish_reason": "stop", "answer": {"type": "noul", "noul": 0}, "latency_s": 0.83921691699652}
