{
  "batch_id": "2026-05-17-batch-01",
  "suite_id": "ofone-v0.5-three-arm-evaluation",
  "label": "Initial five-case benchmark execution plan",
  "status": "in_progress",
  "purpose": "Freeze the first executable benchmark batch before running model comparisons, so later outputs can be audited against predeclared cases, prompts, review criteria, and release guards.",
  "case_ids": [
    "case-strategic-gated-diligence-001",
    "case-regulated-wastewater-market-entry-001",
    "case-formal-proof-search-001",
    "case-public-sector-ai-policy-audit-001",
    "case-scientific-mechanism-check-001"
  ],
  "execution_matrix_file": "benchmarks/runs/2026-05-17-batch-01/execution-matrix.json",
  "arms": [
    {
      "arm_id": "direct_answer",
      "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/direct_answer.md",
      "required_outputs": [
        "answer",
        "confidence_or_uncertainty",
        "source_or_gap_notes"
      ]
    },
    {
      "arm_id": "light_structured",
      "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/light_structured.md",
      "required_outputs": [
        "structured_answer",
        "risks_or_unknowns",
        "recommendation"
      ]
    },
    {
      "arm_id": "full_ofone",
      "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
      "required_outputs": [
        "artifact_json",
        "validator_result",
        "rendering",
        "patch_report_if_update_case"
      ]
    }
  ],
  "model_plan": {
    "runs_per_case_per_arm": 3,
    "model_families": [
      {
        "family_id": "frontier_reasoning",
        "target": "GPT-5.5 Pro or current ChatGPT Pro Deep Research-equivalent reviewer available to the operator",
        "status": "in_progress"
      },
      {
        "family_id": "agentic_coding",
        "target": "Codex-style implementation and validation reviewer available in the repo workflow",
        "status": "completed"
      }
    ],
    "randomization": "Run order should be randomized or rotated by case and arm before execution. Store actual order in run records.",
    "operator_notes": "Do not summarize across arms until raw outputs are saved. Do not let one arm inspect another arm's answer before its own run is complete."
  },
  "review_plan": {
    "rubric": "benchmarks/rubrics/decision-map-rubric.md",
    "review_template": "benchmarks/reviews/2026-05-17-batch-01-review-template.md",
    "blinding": "Hide arm labels from reviewers where practical. If blinding is not possible because full_ofone artifacts are obvious, record that limitation in the review notes.",
    "adjudication_status": "in_progress",
    "independent_review_status": "integrated",
    "independent_review_handoff": "benchmarks/reviews/2026-05-17-batch-01/frontier-independent-review-handoff.md",
    "independent_review_prompt": "research/prompts/2026-05-17-06-ofone-batch01-independent-review.md",
    "independent_review_context": "research/ofone-batch01-independent-review-context.md",
    "independent_review_status_ledger": "research/status/2026-05-17-06-ofone-batch01-independent-review.md",
    "independent_review_url": "https://chatgpt.com/c/6a0a5901-a7fc-83e8-895c-300476365f93",
    "independent_review_launch": {
      "launched_at": "2026-05-17T18:12:12-06:00",
      "model_label": "Latest • 5.5",
      "reasoning_label": "Pro • Extended",
      "deep_research_enabled": true,
      "pasted_context_label": "Pasted text(8).txt",
      "browser_surface": "Computer Use on authenticated Chrome session",
      "launch_proof": [
        "Deep Research plan title visible: Independent OfOne Batch 01 Review",
        "Start clicked after plan generation",
        "Visible status changed to Researching...",
        "Stop research button present"
      ]
    },
    "independent_review_result": "research/results/2026-05-17-06-ofone-batch01-independent-review-result.md",
    "independent_review_harvest": {
      "harvested_at": "2026-05-17T18:29:00-06:00",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (26).md",
      "result_sha256": "sha256:f4f6846dcf80cac78d1eef6217ea40340106ecc35f66764099f7e2aa77a0d5c8",
      "visible_report_metadata": "Research completed in 13m; 15 citations; 5 searches; 17 May; 15 sources",
      "adjudication": "direct_answer accepted; light_structured accepted; full_ofone rejected from aggregate scoring"
    },
    "independent_review_integration": {
      "integrated_at": "2026-05-17T18:45:00-06:00",
      "implemented_changes": [
        "benchmark-case binding for full_ofone artifacts",
        "pre-score compliance gate with auto-reject semantics",
        "immutable validator and patch artifact capture",
        "semantic-fidelity review fields",
        "excluded-run log and matrix state semantics"
      ],
      "blocker_state": "critical Run 06 workflow blockers implemented locally; broader Batch 01 remains in_progress until more predeclared slots run"
    }
  },
  "results_plan": {
    "summary_file": "benchmarks/results/2026-05-17-batch-01-summary.md",
    "failure_analysis_file": "benchmarks/results/2026-05-17-batch-01-failure-analysis.md",
    "raw_output_dir": "benchmarks/runs/2026-05-17-batch-01/outputs",
    "status": "in_progress",
    "excluded_run_log": "benchmarks/results/2026-05-17-batch-01-excluded-runs.md",
    "checker_attestation_file": "benchmarks/results/2026-05-17-batch-01-checker-attestation.json"
  },
  "release_guard": {
    "superiority_claims_allowed": false,
    "reason": "Batch 01 is still in progress. Run 06 rejected one first-slice agentic_coding full_ofone slot from aggregate scoring, the agentic_coding family is locally complete with one remedial replacement, and frontier_reasoning now has three reviewed strategic repeat-1 slots, three regulated wastewater repeat-1 slots, and one formal proof-search direct-answer repeat-1 slot. The original strategic and regulated wastewater frontier full_ofone artifacts both remain excluded by computed validator failures, with controlled replacement evidence recorded for both.",
    "allowed_claim": "Batch 01 has 52 completed and locally reviewed predeclared run slots: all 45 agentic_coding slots, three frontier_reasoning strategic repeat-1 slots, three regulated wastewater frontier_reasoning repeat-1 slots, and one formal proof-search frontier_reasoning direct-answer repeat-1 slot. Three original run slots are excluded before aggregate scoring, three remedial replacements exist under replace_for_aggregate_only, and no performance comparison is supported."
  }
}
