{
  "matrix_id": "2026-05-17-batch-01-matrix",
  "batch_id": "2026-05-17-batch-01",
  "suite_id": "ofone-v0.5-three-arm-evaluation",
  "status": "in_progress",
  "case_ids": [
    "case-strategic-gated-diligence-001",
    "case-regulated-wastewater-market-entry-001",
    "case-formal-proof-search-001",
    "case-public-sector-ai-policy-audit-001",
    "case-scientific-mechanism-check-001"
  ],
  "arms": [
    "direct_answer",
    "light_structured",
    "full_ofone"
  ],
  "model_families": [
    "frontier_reasoning",
    "agentic_coding"
  ],
  "repeats": [
    1,
    2,
    3
  ],
  "expected_run_count": 90,
  "run_id_template": "{batch_id}__{case_id}__{arm_id}__{model_family}__r{repeat}",
  "raw_output_path_template": "benchmarks/runs/2026-05-17-batch-01/outputs/{run_id}.md",
  "review_path_template": "benchmarks/reviews/2026-05-17-batch-01/{run_id}.md",
  "execution_order_policy": "Rotate case, arm, model family, and repeat order before execution. Store actual order in each raw output header.",
  "completion": {
    "queued": 37,
    "completed": 52,
    "reviewed": 52,
    "excluded": 3,
    "blocked": 1,
    "aggregate_eligible": 52,
    "remedial": 3
  },
  "completed_runs": [
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.md",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": false,
      "pre_score_compliance": {
        "case_fidelity": "fail",
        "required_outputs": "pass",
        "independence": "fail",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Artifact identity is copied from case-strategy-micro-001 and is not bound to case-strategic-gated-diligence-001.",
        "reviewer_notes": "Run 06 independent review found schema-valid is not benchmark-valid; reject before aggregate scoring."
      },
      "semantic_fidelity": {
        "case_binding": "fail",
        "copied_example_risk": "high",
        "evidence_provenance_adequacy": "artifact provenance points to a validated wastewater example, not the benchmark case inputs",
        "artifact_source_identity": "artifact_identity.case_id is case-strategy-micro-001; expected case-strategic-gated-diligence-001"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:4ba12cd353b24ee34e57f28ffb407cceebf1c11c07eece6c1969e30e80782a04",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:8abecdbe35662f35aedc1d5f6eabe4e468e7dfa10e3c48393eccc804e1c83f55"
      },
      "independent_adjudication": {
        "decision": "reject",
        "result_file": "research/results/2026-05-17-06-ofone-batch01-independent-review-result.md",
        "result_sha256": "sha256:f4f6846dcf80cac78d1eef6217ea40340106ecc35f66764099f7e2aa77a0d5c8",
        "reviewer": "ChatGPT Deep Research GPT 5.5 Pro / Run 06",
        "harvested_at": "2026-05-17T18:29:00-06:00",
        "reason": "Rich structure was attached to the wrong case/example artifact."
      },
      "exclusion_reason": "case_fidelity_failure_wrong_artifact_identity",
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "rerun_plan": {
        "required": true,
        "status": "reviewed",
        "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
        "reason": "Original full_ofone slot is benchmark-invalid because the artifact was copied from a different case. Preserve it immutably and create a case-native remedial rerun before broader Batch 01 execution resumes.",
        "aggregate_policy": "replace_for_aggregate_only"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r1",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r1",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with content hashes and explicit measurement/confounder unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the scientific mechanism case and run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.artifact.json",
      "benchmark_trace": {
        "case_id": "case-scientific-mechanism-check-001",
        "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/scientific-mechanism-check.md",
        "case_file_sha256": "sha256:1126bb9ba5d97df4a9e82070d16df5ae4ffa61226d3d98a91b6c121671bb8efc",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:5bae3d6429075001dc0feff579bd382810c9419b50bc4c5055272477bdc77b6d"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:26c27c78e94aaaf3d398b63b7e478efd9ab87fde840b03033fcaed7f34762038",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:41c36cf21c42088323e63dae4abc55d138e3d23ed49c2117ddd6a70ece5b7e55",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.rendering.md",
        "rendering_sha256": "sha256:ebcaebc52c8aea368d36d38ee7ccf1e9ff3e5e41c5ddd2b098e2984d0286c01c"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low_to_medium_same_case_source_backed_pattern",
        "evidence_provenance_adequacy": "generic EPA source-backed evidence plus explicit jurisdiction, influent, partner, and customer unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the regulated wastewater case and run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.artifact.json",
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:8ede0c31fda6746c24b09a2646f4eadc4ed063f822ddc9105abcfa09c01402d2",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:f0439ae2f46fc87e5d6d8db52afbe4b01ee13c1721621d30d8a1225244330205",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.rendering.md",
        "rendering_sha256": "sha256:6de80fa4f33e68bada58951ee841fae9c496e7110aeb59a4c724fdb9a3c333bd"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r1",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r1",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level formal benchmark case evidence with explicit proof certificate, countermodel, candidate lemma, and timeout unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the formal proof-search case and run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.artifact.json",
      "benchmark_trace": {
        "case_id": "case-formal-proof-search-001",
        "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/formal-proof-search.md",
        "case_file_sha256": "sha256:01634155f084b1646ac6930ab1cdc4787575fff4daf285c017271e0e2719e756",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:fc0b94ca224173ad518f3f527356dbe998f29d3eb7ca6d4756e2b343def39c64"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:b8fe1840449b7e740bfcd9dfe4df36967b7a96b2244c6b5465c70fa43b15f891",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:44db4bef99730609280120fa161755b40deb3b247c35d6d66c76853196ee0602",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.rendering.md",
        "rendering_sha256": "sha256:f0dc8f97b13762ec5483b8d01dd1c9c6ca9eea8e5c61aa1c502ffcdb0a759858"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r1",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r1",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level public-sector benchmark case evidence with explicit model, rights, appeal, gate, and review-log unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the public-sector AI policy audit case and run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.artifact.json",
      "benchmark_trace": {
        "case_id": "case-public-sector-ai-policy-audit-001",
        "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/public-sector-ai-policy-audit.md",
        "case_file_sha256": "sha256:eafc104563df8651821c28617ce84dc0c43119bd2c09d3fedd2e5fcba52f3178",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:171b27fd434b9cf2dc349fb0c2906778e83caf37a483741f5f39e1d067fc39f4"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:746b26a3fb304aa5e08a5f9ae3192a5c8c405e80ff63e33f6c8ffb2f04704e3e",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:a82115da73c8af320a65cf1a6cb629de9b6342239e37e0f6290580c47d0b7194",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.rendering.md",
        "rendering_sha256": "sha256:3a69e6b93de919c50a5b5cf87cefcebcbd0e1c0bf550885b62f8aa61965a01e4"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r2",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r2",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with stable content hashes, chain-of-custody notes, and explicit release-gate unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the strategic gated diligence repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:c04e33e0c7dadd6013a6f710264bdb23e167a9b95d9d42edd45f2486248e6b17",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:7ef16c667d14a2e4aa622372e3a6fbfd745a912894151196a3ac42d9bec3a97f",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:5524f308e9456dab3c751fb98cde6461ae17d0d1e1f7e26960604dff90f56fa2"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r2",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r2",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with stable content hashes, chain-of-custody notes, and explicit measurement/confounder gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the scientific mechanism repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-scientific-mechanism-check-001",
        "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/scientific-mechanism-check.md",
        "case_file_sha256": "sha256:1126bb9ba5d97df4a9e82070d16df5ae4ffa61226d3d98a91b6c121671bb8efc",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:5bae3d6429075001dc0feff579bd382810c9419b50bc4c5055272477bdc77b6d"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:c47e955d4b1287621c3ddfdc962395a38f3f3ddce83a74bd685c47cded0be351",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:eff1d5c2827d9e4bb1264bc66529476aa6ad0c20e0dfe89bf4afa66aed1b437e",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:a6acfe15047399bdea022dca8d641e291b8aace8bb05017121d936edcebaaa28"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r2",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r2",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "generic public EPA source identity plus scenario-level benchmark unknowns with stable content hashes, chain-of-custody notes, and explicit launch gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the regulated wastewater repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:a1984569f62829fc7c4c7999e440718659140fb9bff4f5948d9eee407740a587",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:83e31e6be1b2d580793b1137ea0e8116ec808aaafd289daf22621b3d81838d76",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:4ea403c1ed03bedb99da6db3e66328925b3fa3c251ef5642e723635a0dc1e9ed"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r2",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r2",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level formal benchmark evidence with stable content hashes, chain-of-custody notes, and explicit proof/countermodel gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the formal proof-search repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-formal-proof-search-001",
        "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/formal-proof-search.md",
        "case_file_sha256": "sha256:01634155f084b1646ac6930ab1cdc4787575fff4daf285c017271e0e2719e756",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:fc0b94ca224173ad518f3f527356dbe998f29d3eb7ca6d4756e2b343def39c64"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:f4bcbc7b386060cd1e128d468f3d79c07aa04abe2f34fbdb7b5b8ecd2a98eac7",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:0fd2c26dfbfc7981e490300036592df95d5d664efcb2455e136015f06acdd3b9",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:c0100c398ee5caf55d04ed9999ad997d69c212261dfdb9ed2d51428674dd8127"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r2",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for the direct text baseline",
        "artifact_source_identity": "raw Markdown output is bound to the public-sector AI policy audit repeat-2 benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r2",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for the lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the public-sector AI policy audit repeat-2 benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level public-sector benchmark case evidence with explicit model, rights, appeal, gate, and review-log unknowns for repeat 2",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the public-sector AI policy audit repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-public-sector-ai-policy-audit-001",
        "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/public-sector-ai-policy-audit.md",
        "case_file_sha256": "sha256:eafc104563df8651821c28617ce84dc0c43119bd2c09d3fedd2e5fcba52f3178",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:171b27fd434b9cf2dc349fb0c2906778e83caf37a483741f5f39e1d067fc39f4"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:c8fbc91c2a469f7a45296358e105875b9918baf848f9d3387caa328c7198bb42",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:3672e7db2a274588f46ac3b39b5203ccdefad8a835503c333846c3d3939924c2",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:1875139d39dece136dfa02c24012919c19dbdb85805df550a830d54fb8609c68"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r3",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r3",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with stable content hashes, chain-of-custody notes, and explicit release-gate unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the strategic gated diligence repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:11ea3e975b056fab1b3ff552bbda265400868aa4a69d0c18d23512692d81d1d3",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:a4d8ddea55c2ab8225d5f868447f5bc0cfefe2ec85a6c149b8b9228ab24c5313",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:e2a014e1b518a4e8a605156112829f1c56bad13e09384e81e8fec1606c2ee5e1"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r3",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r3",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with stable content hashes, chain-of-custody notes, and explicit measurement/confounder gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the scientific mechanism repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-scientific-mechanism-check-001",
        "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/scientific-mechanism-check.md",
        "case_file_sha256": "sha256:1126bb9ba5d97df4a9e82070d16df5ae4ffa61226d3d98a91b6c121671bb8efc",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:5bae3d6429075001dc0feff579bd382810c9419b50bc4c5055272477bdc77b6d"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:9f41d603cef0b80d479f5f1d4b54589c55e234fcd5e03909e6dee74530c7450f",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:7b26cdac7b856d044b7d2a1d5c14f3eb0d6c4d3e054ac153dcff1fb7d2fc3ddc",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:59cdb6490bdbc2d540856a89e503fa038cab98c137324f50b059103307b4243d"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r3",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r3",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "generic public EPA source identity plus scenario-level benchmark unknowns with stable content hashes, chain-of-custody notes, and explicit launch gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the regulated wastewater repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:c2347a2f059985ef501cd98743faef1067ed1d63cfc4ff62771a1af8be72d3e2",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:00b0dc405ad8ab983d0eee5bfe4dbdc20fc543a7f5290dc7ddf9202c22417c51",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:efcc627dedbbf1d377648c99cb3e1912c0e273c68f319abbc1a1d7392071fe2b"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r3",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r3",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level formal benchmark evidence with stable content hashes, chain-of-custody notes, and explicit proof/countermodel gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the formal proof-search repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-formal-proof-search-001",
        "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/formal-proof-search.md",
        "case_file_sha256": "sha256:01634155f084b1646ac6930ab1cdc4787575fff4daf285c017271e0e2719e756",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:fc0b94ca224173ad518f3f527356dbe998f29d3eb7ca6d4756e2b343def39c64"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:2b0254c61a1559d3c2ff5ba77122a8c64b6fcaedcae14c22d8b0baed5483b2a2",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:570e62bec506e91798b49eabb4e828166c9d3b8ce604a6cace9252445e71e5db",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:061c9a215b7209c2b6b575070524d9af34dec122655ab2cf790f553c017aafdb"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r3",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for the direct text baseline",
        "artifact_source_identity": "raw Markdown output is bound to the public-sector AI policy audit repeat-3 benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r3",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for the lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the public-sector AI policy audit repeat-3 benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level public-sector benchmark case evidence with explicit model, rights, appeal, stakeholder, gate, and review-log unknowns for repeat 3",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the public-sector AI policy audit repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-public-sector-ai-policy-audit-001",
        "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/public-sector-ai-policy-audit.md",
        "case_file_sha256": "sha256:eafc104563df8651821c28617ce84dc0c43119bd2c09d3fedd2e5fcba52f3178",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:171b27fd434b9cf2dc349fb0c2906778e83caf37a483741f5f39e1d067fc39f4"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:6387044d914a71bc4bc8ba155a2988aaa916c00e22c5c5f6790b02b671358bc9",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:6af11990c03b2a2ad98080f1b8b8cf1facc637aa71b2f0d9f50e2aa43dfee408",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:dec1fa1ea1a57176470bf719896c3c5fe02bd05fd720b789006f63cd11a671e9"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__frontier_reasoning__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "direct_answer",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "status": "completed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:1647be401efc3069e61c763ed619e7c8cd29c3f212cb1e993d17457a7c4e44e2",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (32).md",
      "harvested_at": "2026-05-20T21:05:00-06:00",
      "visible_report_metadata": "Research completed in 17m; 6 citations; 81 searches",
      "conversation_url": "https://chatgpt.com/c/6a0e3e09-fd6c-83e8-a914-36445d70d090",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier direct-answer run is harvested from a completed ChatGPT Deep Research report and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "public launch/risk-governance sources used for general support; missing case-specific evidence remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-strategic-gated-diligence-001"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__frontier_reasoning__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "light_structured",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:f70a55a9dee97cebc5d0138f57ef887c948dabc452fb2071dec4624fd13b55e6",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (33).md",
      "harvested_at": "2026-05-20T21:43:46-06:00",
      "visible_report_metadata": "Completed report visible with title `Benchmark Raw Output`; run metadata shows Status: `completed`; export to Markdown succeeded.",
      "conversation_url": "https://chatgpt.com/c/6a0e7bcd-43b0-83e8-9a92-5195521c42fe",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier light-structured run is harvested from a completed ChatGPT Deep Research report and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "public launch, deployment, and risk-governance sources used for general support; missing case-specific evidence remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-strategic-gated-diligence-001"
      },
      "status": "completed"
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:227ffdc008b9db9c0facac6de0303d1fd1575f295e8f82da91ce7078014b555a",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (34).md",
      "harvested_at": "2026-05-20T22:29:37-06:00",
      "visible_report_metadata": "Research completed in 18m; 5 citations; 9 searches; report title `Benchmark Raw Output`; run metadata shows Status: `completed`.",
      "conversation_url": "https://chatgpt.com/c/6a0e8476-9f6c-83e8-b201-ff3f97fae18b",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.md",
      "aggregate_eligible": false,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "fail",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Computed local validator failed semantic graph checks and contradicted the artifact self-attested validator_result.passed=true.",
        "reviewer_notes": "Run is case-bound and harvested, but full-OfOne outputs must pass executable local validation before aggregate eligibility."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence plus protocol/validation-model hashes; missing operating facts remain explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, artifact JSON, validator JSON, rendering, patch report, and review file all identify the benchmark run and case-strategic-gated-diligence-001"
      },
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.validator.json",
        "validator_sha256": "sha256:0fd886b42a51dad1fd1d2e3a5925cf15356a0a7215f9d846c849f417ed288f53",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.patch.json",
        "patch_sha256": "sha256:87603090afe09e772cd1af0bc9ea44a5a9652fbecb71e6121a414b0d8deb9a5c",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.rendering.md",
        "rendering_sha256": "sha256:df814569365d29b8485bc2aef3ee5ef0317bc2979f42f1fb3ab6b7db2bdc4483"
      },
      "status": "completed"
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__frontier_reasoning__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "direct_answer",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:afc97b5a4ba4f92eaa8c1c41470e56016ba71dbd9fa7a3a5315770605db80147",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (40).md",
      "harvested_at": "2026-05-21T03:58:46-06:00",
      "visible_report_metadata": "Research completed in 12m; 16 citations; 319 searches; 21 May; 16 sources; title Benchmark Raw Output; Status: completed",
      "conversation_url": "https://chatgpt.com/c/6a0ed3db-cccc-83e8-b84c-b3b1cb7b0bfa",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier direct-answer run is harvested from a completed ChatGPT Deep Research report, answers the regulated wastewater case, and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "primary public regulatory sources are cited for permitting, pretreatment, reuse, operator, residuals, and compliance claims; missing case-specific evidence remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-regulated-wastewater-market-entry-001"
      },
      "status": "completed"
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__frontier_reasoning__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "light_structured",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:cf4eb63a53dc211ea09bb51030e3771129207cbe4e81f9c231388299fbeb9042",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (41).md",
      "harvested_at": "2026-05-21T04:36:36-06:00",
      "visible_report_metadata": "Research completed in 22m; 15 citations; 236 searches; 21 May; 15 sources; title Benchmark Raw Output; Status: completed",
      "conversation_url": "https://chatgpt.com/c/6a0eda60-dc18-83e8-a888-e9e8ac1ab1fe",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier light-structured run is harvested from a completed ChatGPT Deep Research report, answers the regulated wastewater case, and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "primary public regulatory sources are cited for permitting, reuse, residuals, PFAS, partner, customer-commitment, and compliance claims; missing case-specific evidence remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-regulated-wastewater-market-entry-001"
      },
      "status": "completed"
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:047dbd2e832eea80067ef3865ab056158dc9bb5c67b391b809783afd89170a9e",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (42).md",
      "harvested_at": "2026-05-21T05:45:37-06:00",
      "visible_report_metadata": "Research completed in 45m; 5 citations; 22 searches; 21 May; 5 sources; title Benchmark Raw Output; Status: completed",
      "conversation_url": "https://chatgpt.com/c/6a0ee4ad-8854-83e8-866e-f671c12880da",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.md",
      "aggregate_eligible": false,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "fail",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Computed local validator failed semantic graph checks despite case-bound artifact identity and required package artifacts.",
        "reviewer_notes": "Run is case-bound and harvested, but full-OfOne outputs must pass executable local validation before aggregate eligibility."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "source-backed regulatory, permitting, reuse, residuals, PFAS, partner, customer-commitment, and compliance evidence is separated from missing case-specific proof; computed semantic validation still failed",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, artifact JSON, validator JSON, rendering, patch report, and review file all identify the benchmark run and case-regulated-wastewater-market-entry-001"
      },
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.validator.json",
        "validator_sha256": "sha256:a5418d11c048b22749b69d4f76b47f6a0dfe3a2bb823932a5d73750a3b21f2c1",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.patch.json",
        "patch_sha256": "sha256:49c41e22d98e027e5c0dc70ff733669e9be0383de98e16e06f22709618afe9e5",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.rendering.md",
        "rendering_sha256": "sha256:55b6ab910eb831fd9869a4b969c804658fb7710f1e5fc5af782492349552db45"
      },
      "status": "completed",
      "rerun_plan": {
        "required": true,
        "status": "reviewed",
        "rerun_of": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
        "reason": "Frontier full-OfOne regulated wastewater repeat-1 artifact is preserved immutably as excluded because computed local validation failed semantic graph checks. Controlled Mode A rerun 1 is validator-valid replacement evidence only.",
        "aggregate_policy": "replace_for_aggregate_only",
        "latest_successful_rerun": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__frontier_reasoning__r1",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "direct_answer",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:16ee781550c471e01b92ae4f577a04de8d54a6be0ec27a6dab34d547dcbef784",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (43).md",
      "harvested_at": "2026-05-21T08:50:37-06:00",
      "visible_report_metadata": "Research completed in 10m; 8 citations; 101 searches; 21 May; title Benchmark Raw Output; Status: completed",
      "conversation_url": "https://chatgpt.com/c/6a0f0a85-c75c-83e8-b0d0-4c15a041cb7b",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier direct-answer run is harvested from a completed ChatGPT Deep Research report, answers the formal proof-search case, and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "formal-methods/tool documentation supports proof obligations, countermodel search, bounded-search limits, stale proofs, and unsat/vacuity checks; missing object-level theory remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-formal-proof-search-001"
      },
      "status": "completed"
    }
  ],
  "reviewed_runs": [
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.md",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": false,
      "pre_score_compliance": {
        "case_fidelity": "fail",
        "required_outputs": "pass",
        "independence": "fail",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Artifact identity is copied from case-strategy-micro-001 and is not bound to case-strategic-gated-diligence-001.",
        "reviewer_notes": "Run 06 independent review found schema-valid is not benchmark-valid; reject before aggregate scoring."
      },
      "semantic_fidelity": {
        "case_binding": "fail",
        "copied_example_risk": "high",
        "evidence_provenance_adequacy": "artifact provenance points to a validated wastewater example, not the benchmark case inputs",
        "artifact_source_identity": "artifact_identity.case_id is case-strategy-micro-001; expected case-strategic-gated-diligence-001"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:4ba12cd353b24ee34e57f28ffb407cceebf1c11c07eece6c1969e30e80782a04",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:8abecdbe35662f35aedc1d5f6eabe4e468e7dfa10e3c48393eccc804e1c83f55"
      },
      "independent_adjudication": {
        "decision": "reject",
        "result_file": "research/results/2026-05-17-06-ofone-batch01-independent-review-result.md",
        "result_sha256": "sha256:f4f6846dcf80cac78d1eef6217ea40340106ecc35f66764099f7e2aa77a0d5c8",
        "reviewer": "ChatGPT Deep Research GPT 5.5 Pro / Run 06",
        "harvested_at": "2026-05-17T18:29:00-06:00",
        "reason": "Rich structure was attached to the wrong case/example artifact."
      },
      "exclusion_reason": "case_fidelity_failure_wrong_artifact_identity",
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "rerun_plan": {
        "required": true,
        "status": "reviewed",
        "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
        "reason": "Original full_ofone slot is benchmark-invalid because the artifact was copied from a different case. Preserve it immutably and create a case-native remedial rerun before broader Batch 01 execution resumes.",
        "aggregate_policy": "replace_for_aggregate_only"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r1",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r1",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with content hashes and explicit measurement/confounder unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the scientific mechanism case and run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.artifact.json",
      "benchmark_trace": {
        "case_id": "case-scientific-mechanism-check-001",
        "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/scientific-mechanism-check.md",
        "case_file_sha256": "sha256:1126bb9ba5d97df4a9e82070d16df5ae4ffa61226d3d98a91b6c121671bb8efc",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:5bae3d6429075001dc0feff579bd382810c9419b50bc4c5055272477bdc77b6d"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:26c27c78e94aaaf3d398b63b7e478efd9ab87fde840b03033fcaed7f34762038",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:41c36cf21c42088323e63dae4abc55d138e3d23ed49c2117ddd6a70ece5b7e55",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r1.rendering.md",
        "rendering_sha256": "sha256:ebcaebc52c8aea368d36d38ee7ccf1e9ff3e5e41c5ddd2b098e2984d0286c01c"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low_to_medium_same_case_source_backed_pattern",
        "evidence_provenance_adequacy": "generic EPA source-backed evidence plus explicit jurisdiction, influent, partner, and customer unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the regulated wastewater case and run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.artifact.json",
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:8ede0c31fda6746c24b09a2646f4eadc4ed063f822ddc9105abcfa09c01402d2",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:f0439ae2f46fc87e5d6d8db52afbe4b01ee13c1721621d30d8a1225244330205",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r1.rendering.md",
        "rendering_sha256": "sha256:6de80fa4f33e68bada58951ee841fae9c496e7110aeb59a4c724fdb9a3c333bd"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r1",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r1",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level formal benchmark case evidence with explicit proof certificate, countermodel, candidate lemma, and timeout unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the formal proof-search case and run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.artifact.json",
      "benchmark_trace": {
        "case_id": "case-formal-proof-search-001",
        "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/formal-proof-search.md",
        "case_file_sha256": "sha256:01634155f084b1646ac6930ab1cdc4787575fff4daf285c017271e0e2719e756",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:fc0b94ca224173ad518f3f527356dbe998f29d3eb7ca6d4756e2b343def39c64"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:b8fe1840449b7e740bfcd9dfe4df36967b7a96b2244c6b5465c70fa43b15f891",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:44db4bef99730609280120fa161755b40deb3b247c35d6d66c76853196ee0602",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r1.rendering.md",
        "rendering_sha256": "sha256:f0dc8f97b13762ec5483b8d01dd1c9c6ca9eea8e5c61aa1c502ffcdb0a759858"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r1",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r1",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence; acceptable for lightweight text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level public-sector benchmark case evidence with explicit model, rights, appeal, gate, and review-log unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the public-sector AI policy audit case and run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.artifact.json",
      "benchmark_trace": {
        "case_id": "case-public-sector-ai-policy-audit-001",
        "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/public-sector-ai-policy-audit.md",
        "case_file_sha256": "sha256:eafc104563df8651821c28617ce84dc0c43119bd2c09d3fedd2e5fcba52f3178",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:171b27fd434b9cf2dc349fb0c2906778e83caf37a483741f5f39e1d067fc39f4"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:746b26a3fb304aa5e08a5f9ae3192a5c8c405e80ff63e33f6c8ffb2f04704e3e",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:a82115da73c8af320a65cf1a6cb629de9b6342239e37e0f6290580c47d0b7194",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r1.rendering.md",
        "rendering_sha256": "sha256:3a69e6b93de919c50a5b5cf87cefcebcbd0e1c0bf550885b62f8aa61965a01e4"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r2",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r2",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with stable content hashes, chain-of-custody notes, and explicit release-gate unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the strategic gated diligence repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:c04e33e0c7dadd6013a6f710264bdb23e167a9b95d9d42edd45f2486248e6b17",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:7ef16c667d14a2e4aa622372e3a6fbfd745a912894151196a3ac42d9bec3a97f",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:5524f308e9456dab3c751fb98cde6461ae17d0d1e1f7e26960604dff90f56fa2"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r2",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r2",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with stable content hashes, chain-of-custody notes, and explicit measurement/confounder gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the scientific mechanism repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-scientific-mechanism-check-001",
        "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/scientific-mechanism-check.md",
        "case_file_sha256": "sha256:1126bb9ba5d97df4a9e82070d16df5ae4ffa61226d3d98a91b6c121671bb8efc",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:5bae3d6429075001dc0feff579bd382810c9419b50bc4c5055272477bdc77b6d"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:c47e955d4b1287621c3ddfdc962395a38f3f3ddce83a74bd685c47cded0be351",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:eff1d5c2827d9e4bb1264bc66529476aa6ad0c20e0dfe89bf4afa66aed1b437e",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:a6acfe15047399bdea022dca8d641e291b8aace8bb05017121d936edcebaaa28"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r2",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r2",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "generic public EPA source identity plus scenario-level benchmark unknowns with stable content hashes, chain-of-custody notes, and explicit launch gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the regulated wastewater repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:a1984569f62829fc7c4c7999e440718659140fb9bff4f5948d9eee407740a587",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:83e31e6be1b2d580793b1137ea0e8116ec808aaafd289daf22621b3d81838d76",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:4ea403c1ed03bedb99da6db3e66328925b3fa3c251ef5642e723635a0dc1e9ed"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r2",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r2",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level formal benchmark evidence with stable content hashes, chain-of-custody notes, and explicit proof/countermodel gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the formal proof-search repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-formal-proof-search-001",
        "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/formal-proof-search.md",
        "case_file_sha256": "sha256:01634155f084b1646ac6930ab1cdc4787575fff4daf285c017271e0e2719e756",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:fc0b94ca224173ad518f3f527356dbe998f29d3eb7ca6d4756e2b343def39c64"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:f4bcbc7b386060cd1e128d468f3d79c07aa04abe2f34fbdb7b5b8ecd2a98eac7",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:0fd2c26dfbfc7981e490300036592df95d5d664efcb2455e136015f06acdd3b9",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:c0100c398ee5caf55d04ed9999ad997d69c212261dfdb9ed2d51428674dd8127"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r2",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for the direct text baseline",
        "artifact_source_identity": "raw Markdown output is bound to the public-sector AI policy audit repeat-2 benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r2",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for the lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the public-sector AI policy audit repeat-2 benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 2,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level public-sector benchmark case evidence with explicit model, rights, appeal, gate, and review-log unknowns for repeat 2",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the public-sector AI policy audit repeat-2 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.artifact.json",
      "benchmark_trace": {
        "case_id": "case-public-sector-ai-policy-audit-001",
        "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2",
        "case_file": "benchmarks/cases/public-sector-ai-policy-audit.md",
        "case_file_sha256": "sha256:eafc104563df8651821c28617ce84dc0c43119bd2c09d3fedd2e5fcba52f3178",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:171b27fd434b9cf2dc349fb0c2906778e83caf37a483741f5f39e1d067fc39f4"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.validator.json",
        "validator_sha256": "sha256:c8fbc91c2a469f7a45296358e105875b9918baf848f9d3387caa328c7198bb42",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.patch.json",
        "patch_sha256": "sha256:3672e7db2a274588f46ac3b39b5203ccdefad8a835503c333846c3d3939924c2",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r2.rendering.md",
        "rendering_sha256": "sha256:1875139d39dece136dfa02c24012919c19dbdb85805df550a830d54fb8609c68"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r3",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r3",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with stable content hashes, chain-of-custody notes, and explicit release-gate unknowns",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the strategic gated diligence repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:11ea3e975b056fab1b3ff552bbda265400868aa4a69d0c18d23512692d81d1d3",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:a4d8ddea55c2ab8225d5f868447f5bc0cfefe2ec85a6c149b8b9228ab24c5313",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:e2a014e1b518a4e8a605156112829f1c56bad13e09384e81e8fec1606c2ee5e1"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r3",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r3",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3",
      "case_id": "case-scientific-mechanism-check-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with stable content hashes, chain-of-custody notes, and explicit measurement/confounder gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the scientific mechanism repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-scientific-mechanism-check-001",
        "run_id": "2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/scientific-mechanism-check.md",
        "case_file_sha256": "sha256:1126bb9ba5d97df4a9e82070d16df5ae4ffa61226d3d98a91b6c121671bb8efc",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:5bae3d6429075001dc0feff579bd382810c9419b50bc4c5055272477bdc77b6d"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:9f41d603cef0b80d479f5f1d4b54589c55e234fcd5e03909e6dee74530c7450f",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:7b26cdac7b856d044b7d2a1d5c14f3eb0d6c4d3e054ac153dcff1fb7d2fc3ddc",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-scientific-mechanism-check-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:59cdb6490bdbc2d540856a89e503fa038cab98c137324f50b059103307b4243d"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r3",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r3",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "generic public EPA source identity plus scenario-level benchmark unknowns with stable content hashes, chain-of-custody notes, and explicit launch gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the regulated wastewater repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:c2347a2f059985ef501cd98743faef1067ed1d63cfc4ff62771a1af8be72d3e2",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:00b0dc405ad8ab983d0eee5bfe4dbdc20fc543a7f5290dc7ddf9202c22417c51",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:efcc627dedbbf1d377648c99cb3e1912c0e273c68f319abbc1a1d7392071fe2b"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r3",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for baseline text arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r3",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level formal benchmark evidence with stable content hashes, chain-of-custody notes, and explicit proof/countermodel gates",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the formal proof-search repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-formal-proof-search-001",
        "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/formal-proof-search.md",
        "case_file_sha256": "sha256:01634155f084b1646ac6930ab1cdc4787575fff4daf285c017271e0e2719e756",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:fc0b94ca224173ad518f3f527356dbe998f29d3eb7ca6d4756e2b343def39c64"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:2b0254c61a1559d3c2ff5ba77122a8c64b6fcaedcae14c22d8b0baed5483b2a2",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:570e62bec506e91798b49eabb4e828166c9d3b8ce604a6cace9252445e71e5db",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:061c9a215b7209c2b6b575070524d9af34dec122655ab2cf790f553c017aafdb"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r3",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "direct_answer",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__direct_answer__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for the direct text baseline",
        "artifact_source_identity": "raw Markdown output is bound to the public-sector AI policy audit repeat-3 benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r3",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "light_structured",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__light_structured__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level only; acceptable for the lightweight structured arm",
        "artifact_source_identity": "raw Markdown output is bound to the public-sector AI policy audit repeat-3 benchmark run metadata."
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3",
      "case_id": "case-public-sector-ai-policy-audit-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 3,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.md",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level public-sector benchmark case evidence with explicit model, rights, appeal, stakeholder, gate, and review-log unknowns for repeat 3",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the public-sector AI policy audit repeat-3 run."
      },
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.artifact.json",
      "benchmark_trace": {
        "case_id": "case-public-sector-ai-policy-audit-001",
        "run_id": "2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3",
        "case_file": "benchmarks/cases/public-sector-ai-policy-audit.md",
        "case_file_sha256": "sha256:eafc104563df8651821c28617ce84dc0c43119bd2c09d3fedd2e5fcba52f3178",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:171b27fd434b9cf2dc349fb0c2906778e83caf37a483741f5f39e1d067fc39f4"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.validator.json",
        "validator_sha256": "sha256:6387044d914a71bc4bc8ba155a2988aaa916c00e22c5c5f6790b02b671358bc9",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.patch.json",
        "patch_sha256": "sha256:6af11990c03b2a2ad98080f1b8b8cf1facc637aa71b2f0d9f50e2aa43dfee408",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-public-sector-ai-policy-audit-001__full_ofone__agentic_coding__r3.rendering.md",
        "rendering_sha256": "sha256:dec1fa1ea1a57176470bf719896c3c5fe02bd05fd720b789006f63cd11a671e9"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__frontier_reasoning__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "direct_answer",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:1647be401efc3069e61c763ed619e7c8cd29c3f212cb1e993d17457a7c4e44e2",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (32).md",
      "harvested_at": "2026-05-20T21:05:00-06:00",
      "visible_report_metadata": "Research completed in 17m; 6 citations; 81 searches",
      "conversation_url": "https://chatgpt.com/c/6a0e3e09-fd6c-83e8-a914-36445d70d090",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__direct_answer__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier direct-answer run is harvested from a completed ChatGPT Deep Research report and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "public launch/risk-governance sources used for general support; missing case-specific evidence remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-strategic-gated-diligence-001"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__frontier_reasoning__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "light_structured",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:f70a55a9dee97cebc5d0138f57ef887c948dabc452fb2071dec4624fd13b55e6",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (33).md",
      "harvested_at": "2026-05-20T21:43:46-06:00",
      "visible_report_metadata": "Completed report visible with title `Benchmark Raw Output`; run metadata shows Status: `completed`; export to Markdown succeeded.",
      "conversation_url": "https://chatgpt.com/c/6a0e7bcd-43b0-83e8-9a92-5195521c42fe",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__light_structured__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier light-structured run is harvested from a completed ChatGPT Deep Research report and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "public launch, deployment, and risk-governance sources used for general support; missing case-specific evidence remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-strategic-gated-diligence-001"
      },
      "status": "reviewed"
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:227ffdc008b9db9c0facac6de0303d1fd1575f295e8f82da91ce7078014b555a",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (34).md",
      "harvested_at": "2026-05-20T22:29:37-06:00",
      "visible_report_metadata": "Research completed in 18m; 5 citations; 9 searches; report title `Benchmark Raw Output`; run metadata shows Status: `completed`.",
      "conversation_url": "https://chatgpt.com/c/6a0e8476-9f6c-83e8-b201-ff3f97fae18b",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.md",
      "aggregate_eligible": false,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "fail",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Computed local validator failed semantic graph checks and contradicted the artifact self-attested validator_result.passed=true.",
        "reviewer_notes": "Run is case-bound and harvested, but full-OfOne outputs must pass executable local validation before aggregate eligibility."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence plus protocol/validation-model hashes; missing operating facts remain explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, artifact JSON, validator JSON, rendering, patch report, and review file all identify the benchmark run and case-strategic-gated-diligence-001"
      },
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.validator.json",
        "validator_sha256": "sha256:0fd886b42a51dad1fd1d2e3a5925cf15356a0a7215f9d846c849f417ed288f53",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.patch.json",
        "patch_sha256": "sha256:87603090afe09e772cd1af0bc9ea44a5a9652fbecb71e6121a414b0d8deb9a5c",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.rendering.md",
        "rendering_sha256": "sha256:df814569365d29b8485bc2aef3ee5ef0317bc2979f42f1fb3ab6b7db2bdc4483"
      },
      "status": "reviewed"
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__frontier_reasoning__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "direct_answer",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:afc97b5a4ba4f92eaa8c1c41470e56016ba71dbd9fa7a3a5315770605db80147",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (40).md",
      "harvested_at": "2026-05-21T03:58:46-06:00",
      "visible_report_metadata": "Research completed in 12m; 16 citations; 319 searches; 21 May; 16 sources; title Benchmark Raw Output; Status: completed",
      "conversation_url": "https://chatgpt.com/c/6a0ed3db-cccc-83e8-b84c-b3b1cb7b0bfa",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__direct_answer__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier direct-answer run is harvested from a completed ChatGPT Deep Research report, answers the regulated wastewater case, and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "primary public regulatory sources are cited for permitting, pretreatment, reuse, operator, residuals, and compliance claims; missing case-specific evidence remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-regulated-wastewater-market-entry-001"
      },
      "status": "reviewed"
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__frontier_reasoning__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "light_structured",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:cf4eb63a53dc211ea09bb51030e3771129207cbe4e81f9c231388299fbeb9042",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (41).md",
      "harvested_at": "2026-05-21T04:36:36-06:00",
      "visible_report_metadata": "Research completed in 22m; 15 citations; 236 searches; 21 May; 15 sources; title Benchmark Raw Output; Status: completed",
      "conversation_url": "https://chatgpt.com/c/6a0eda60-dc18-83e8-a888-e9e8ac1ab1fe",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__light_structured__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier light-structured run is harvested from a completed ChatGPT Deep Research report, answers the regulated wastewater case, and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "primary public regulatory sources are cited for permitting, reuse, residuals, PFAS, partner, customer-commitment, and compliance claims; missing case-specific evidence remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-regulated-wastewater-market-entry-001"
      },
      "status": "reviewed"
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:047dbd2e832eea80067ef3865ab056158dc9bb5c67b391b809783afd89170a9e",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (42).md",
      "harvested_at": "2026-05-21T05:45:37-06:00",
      "visible_report_metadata": "Research completed in 45m; 5 citations; 22 searches; 21 May; 5 sources; title Benchmark Raw Output; Status: completed",
      "conversation_url": "https://chatgpt.com/c/6a0ee4ad-8854-83e8-866e-f671c12880da",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.md",
      "aggregate_eligible": false,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "fail",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Computed local validator failed semantic graph checks despite case-bound artifact identity and required package artifacts.",
        "reviewer_notes": "Run is case-bound and harvested, but full-OfOne outputs must pass executable local validation before aggregate eligibility."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "source-backed regulatory, permitting, reuse, residuals, PFAS, partner, customer-commitment, and compliance evidence is separated from missing case-specific proof; computed semantic validation still failed",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, artifact JSON, validator JSON, rendering, patch report, and review file all identify the benchmark run and case-regulated-wastewater-market-entry-001"
      },
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.validator.json",
        "validator_sha256": "sha256:a5418d11c048b22749b69d4f76b47f6a0dfe3a2bb823932a5d73750a3b21f2c1",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.patch.json",
        "patch_sha256": "sha256:49c41e22d98e027e5c0dc70ff733669e9be0383de98e16e06f22709618afe9e5",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.rendering.md",
        "rendering_sha256": "sha256:55b6ab910eb831fd9869a4b969c804658fb7710f1e5fc5af782492349552db45"
      },
      "status": "reviewed",
      "rerun_plan": {
        "required": true,
        "status": "reviewed",
        "rerun_of": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
        "reason": "Frontier full-OfOne regulated wastewater repeat-1 artifact is preserved immutably as excluded because computed local validation failed semantic graph checks. Controlled Mode A rerun 1 is validator-valid replacement evidence only.",
        "aggregate_policy": "replace_for_aggregate_only",
        "latest_successful_rerun": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__frontier_reasoning__r1",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "direct_answer",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:16ee781550c471e01b92ae4f577a04de8d54a6be0ec27a6dab34d547dcbef784",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (43).md",
      "harvested_at": "2026-05-21T08:50:37-06:00",
      "visible_report_metadata": "Research completed in 10m; 8 citations; 101 searches; 21 May; title Benchmark Raw Output; Status: completed",
      "conversation_url": "https://chatgpt.com/c/6a0f0a85-c75c-83e8-b0d0-4c15a041cb7b",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__direct_answer__frontier_reasoning__r1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Pre-score gate passed before metric scoring. Frontier direct-answer run is harvested from a completed ChatGPT Deep Research report, answers the formal proof-search case, and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "formal-methods/tool documentation supports proof obligations, countermodel search, bounded-search limits, stale proofs, and unsat/vacuity checks; missing object-level theory remains explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, review file, and run metadata all identify the benchmark run and case-formal-proof-search-001"
      },
      "status": "reviewed"
    }
  ],
  "release_guard": {
    "superiority_claims_allowed": false,
    "reason": "Batch 01 is still in progress. Run 06 rejected one first-slice agentic_coding full_ofone slot from aggregate scoring, the agentic_coding family is locally complete with one remedial replacement, and frontier_reasoning now has three reviewed strategic repeat-1 slots, three regulated wastewater repeat-1 slots, and one formal proof-search direct-answer repeat-1 slot. Both original frontier full_ofone originals completed but remain excluded by computed validator failures, with controlled replacement evidence recorded for both.",
    "allowed_claim": "Batch 01 has 52 completed and locally reviewed predeclared run slots: all 45 agentic_coding slots, three frontier_reasoning strategic repeat-1 slots, three regulated wastewater frontier_reasoning repeat-1 slots, and one formal proof-search frontier_reasoning direct-answer repeat-1 slot. Three original run slots are excluded before aggregate scoring, three remedial replacements exist, and no performance comparison is supported."
  },
  "excluded_runs": [
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "excluded",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.md",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.md",
      "aggregate_eligible": false,
      "pre_score_compliance": {
        "case_fidelity": "fail",
        "required_outputs": "pass",
        "independence": "fail",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Artifact identity is copied from case-strategy-micro-001 and is not bound to case-strategic-gated-diligence-001.",
        "reviewer_notes": "Run 06 independent review found schema-valid is not benchmark-valid; reject before aggregate scoring."
      },
      "semantic_fidelity": {
        "case_binding": "fail",
        "copied_example_risk": "high",
        "evidence_provenance_adequacy": "artifact provenance points to a validated wastewater example, not the benchmark case inputs",
        "artifact_source_identity": "artifact_identity.case_id is case-strategy-micro-001; expected case-strategic-gated-diligence-001"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.validator.json",
        "validator_sha256": "sha256:4ba12cd353b24ee34e57f28ffb407cceebf1c11c07eece6c1969e30e80782a04",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1.patch.json",
        "patch_sha256": "sha256:8abecdbe35662f35aedc1d5f6eabe4e468e7dfa10e3c48393eccc804e1c83f55"
      },
      "independent_adjudication": {
        "decision": "reject",
        "result_file": "research/results/2026-05-17-06-ofone-batch01-independent-review-result.md",
        "result_sha256": "sha256:f4f6846dcf80cac78d1eef6217ea40340106ecc35f66764099f7e2aa77a0d5c8",
        "reviewer": "ChatGPT Deep Research GPT 5.5 Pro / Run 06",
        "harvested_at": "2026-05-17T18:29:00-06:00",
        "reason": "Rich structure was attached to the wrong case/example artifact."
      },
      "exclusion_reason": "case_fidelity_failure_wrong_artifact_identity",
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "rerun_plan": {
        "required": true,
        "status": "reviewed",
        "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
        "reason": "Original full_ofone slot is benchmark-invalid because the artifact was copied from a different case. Preserve it immutably and create a case-native remedial rerun before broader Batch 01 execution resumes.",
        "aggregate_policy": "replace_for_aggregate_only"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:227ffdc008b9db9c0facac6de0303d1fd1575f295e8f82da91ce7078014b555a",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (34).md",
      "harvested_at": "2026-05-20T22:29:37-06:00",
      "visible_report_metadata": "Research completed in 18m; 5 citations; 9 searches; report title `Benchmark Raw Output`; run metadata shows Status: `completed`.",
      "conversation_url": "https://chatgpt.com/c/6a0e8476-9f6c-83e8-b201-ff3f97fae18b",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.md",
      "aggregate_eligible": false,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "fail",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Computed local validator failed semantic graph checks and contradicted the artifact self-attested validator_result.passed=true.",
        "reviewer_notes": "Run is case-bound and harvested, but full-OfOne outputs must pass executable local validation before aggregate eligibility."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence plus protocol/validation-model hashes; missing operating facts remain explicit",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, artifact JSON, validator JSON, rendering, patch report, and review file all identify the benchmark run and case-strategic-gated-diligence-001"
      },
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.validator.json",
        "validator_sha256": "sha256:0fd886b42a51dad1fd1d2e3a5925cf15356a0a7215f9d846c849f417ed288f53",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.patch.json",
        "patch_sha256": "sha256:87603090afe09e772cd1af0bc9ea44a5a9652fbecb71e6121a414b0d8deb9a5c",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.rendering.md",
        "rendering_sha256": "sha256:df814569365d29b8485bc2aef3ee5ef0317bc2979f42f1fb3ab6b7db2bdc4483"
      },
      "status": "excluded",
      "exclusion_reason": "semantic_validation_failure_computed_validator_failed",
      "independent_adjudication": {
        "decision": "reject",
        "result_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1.md",
        "result_sha256": "sha256:9ea77568013b85edccacede3414d03f8e8a4b61588de06d72b654e9bee6ae83d",
        "reviewer": "local Codex unblinded review plus executable OfOne validator",
        "harvested_at": "2026-05-20T22:29:37-06:00",
        "reason": "Computed validator rejected relation legality and option expected-effect references; artifact self-attested pass contradicts computed validation."
      },
      "rerun_plan": {
        "required": true,
        "status": "reviewed",
        "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
        "reason": "Frontier full-OfOne repeat-1 artifact is preserved immutably as excluded. Four same-shape remedial attempts failed. Controlled Mode A rerun 5 is validator-valid, locally reviewed, and recorded in remedial_runs as replacement evidence for aggregate scoring only.",
        "aggregate_policy": "replace_for_aggregate_only",
        "latest_successful_rerun": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun5"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.md",
      "raw_output_sha256": "sha256:047dbd2e832eea80067ef3865ab056158dc9bb5c67b391b809783afd89170a9e",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (42).md",
      "harvested_at": "2026-05-21T05:45:37-06:00",
      "visible_report_metadata": "Research completed in 45m; 5 citations; 22 searches; 21 May; 5 sources; title Benchmark Raw Output; Status: completed",
      "conversation_url": "https://chatgpt.com/c/6a0ee4ad-8854-83e8-866e-f671c12880da",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.md",
      "aggregate_eligible": false,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "fail",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Computed local validator failed semantic graph checks despite case-bound artifact identity and required package artifacts.",
        "reviewer_notes": "Run is case-bound and harvested, but full-OfOne outputs must pass executable local validation before aggregate eligibility."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "source-backed regulatory, permitting, reuse, residuals, PFAS, partner, customer-commitment, and compliance evidence is separated from missing case-specific proof; computed semantic validation still failed",
        "artifact_source_identity": "raw Markdown output, conversation URL, downloaded export, artifact JSON, validator JSON, rendering, patch report, and review file all identify the benchmark run and case-regulated-wastewater-market-entry-001"
      },
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.validator.json",
        "validator_sha256": "sha256:a5418d11c048b22749b69d4f76b47f6a0dfe3a2bb823932a5d73750a3b21f2c1",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.patch.json",
        "patch_sha256": "sha256:49c41e22d98e027e5c0dc70ff733669e9be0383de98e16e06f22709618afe9e5",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.rendering.md",
        "rendering_sha256": "sha256:55b6ab910eb831fd9869a4b969c804658fb7710f1e5fc5af782492349552db45"
      },
      "status": "excluded",
      "exclusion_reason": "semantic_validation_failure_relation_legality_and_family",
      "independent_adjudication": {
        "decision": "reject",
        "result_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1.md",
        "result_sha256": "sha256:92b54a984eb77efc56964e92f2bd03f83b907e482d936dc085f071f3b6e2e11a",
        "reviewer": "local Codex unblinded review plus executable OfOne validator",
        "harvested_at": "2026-05-21T05:45:37-06:00",
        "reason": "Computed validator rejected illegal option_move-to-unknown updates, illegal claim-to-gate supports, and invalid relation family for gate-to-rendering constrains."
      },
      "rerun_plan": {
        "required": true,
        "status": "reviewed",
        "rerun_of": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
        "reason": "Frontier full-OfOne regulated wastewater repeat-1 artifact is preserved immutably as excluded because computed local validation failed semantic graph checks. Controlled Mode A rerun 1 is validator-valid replacement evidence only.",
        "aggregate_policy": "replace_for_aggregate_only",
        "latest_successful_rerun": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1"
      }
    }
  ],
  "state_semantics": "completed_runs records raw outputs that have completed. reviewed_runs is an overlapping audit state: every reviewed run must also appear in completed_runs. excluded_runs is an overlapping adjudication state for reviewed runs rejected before aggregate scoring. blocked_runs is a non-complete external-observation state: a run may be completed_report_visible in ChatGPT but must not appear in completed_runs, reviewed_runs, excluded_runs, released_runs, or aggregate-eligible counts until native raw Markdown is harvested, locally reviewed, committed, pushed, and Pages-confirmed.",
  "blocked_runs": [
    {
      "run_id": "2026-05-17-batch-01__case-formal-proof-search-001__light_structured__frontier_reasoning__r1",
      "case_id": "case-formal-proof-search-001",
      "arm_id": "light_structured",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "status": "completed_report_visible",
      "aggregate_eligible": false,
      "source_surface": "chrome_extension_plugin",
      "conversation_url": "https://chatgpt.com/c/6a0f1fe5-3494-83e8-9f92-1a2b732c4958",
      "visible_report_metadata": "Research completed in 9m; 6 citations; 120 searches; 21 May; report title Benchmark Raw Output; run metadata Status: completed; correct light_structured run ID visible.",
      "blocker": "Completed report is visible through Chrome extension control, but raw Markdown harvest is blocked because the report body and download control are inside ChatGPT's cross-origin Deep Research sandbox iframe. Browser, Computer Use, coordinate clicking, AppleScript/JXA, generic desktop automation, OCR, and screenshot reconstruction remain disallowed.",
      "expected_raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__frontier_reasoning__r1.md",
      "expected_review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-formal-proof-search-001__light_structured__frontier_reasoning__r1.md",
      "extension_report": "research/deep-research-extension-report.json",
      "manual_recovery_report": "research/deep-research-manual-recovery.json",
      "launch_queue": "research/deep-research-launch-queue.json",
      "extension_payload": "research/deep-research-extension-payloads.json",
      "promotion_gate": {
        "native_markdown_export_required": true,
        "local_review_required": true,
        "queue_payload_report_update_required": true,
        "commit_push_pages_required": true,
        "aggregate_eligible_before_review_publication": false
      }
    }
  ],
  "rerun_policy": {
    "preserve_original_runs": true,
    "rerun_id_template": "{original_run_id}__rerun{rerun_number}",
    "replacement_policy": "Never mutate or delete excluded originals. A remedial rerun creates a new run record linked with rerun_of and can replace the excluded original only for aggregate scoring after independent review.",
    "aggregate_policy": "replace_for_aggregate_only",
    "planned_repeat_policy": "The first remedial rerun repairs the excluded repeat-1 slot and does not consume repeat 2 or repeat 3. Additional failures remain separate excluded records with their own rerun_plan."
  },
  "remedial_runs": [
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1__rerun1",
      "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1",
      "rerun_number": 1,
      "rerun_reason": "Remedial full_ofone rerun after Run 06/Run 07 found and hardened the wrong-case copied-artifact failure mode.",
      "aggregate_policy": "replace_for_aggregate_only",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "agentic_coding",
      "repeat": 1,
      "status": "reviewed",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1__rerun1.md",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1__rerun1.artifact.json",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1__rerun1.md",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Remedial rerun is case-native, carries benchmark trace hash binding, and keeps superiority claims blocked."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case evidence with content hashes and explicit blocking unknown",
        "artifact_source_identity": "artifact_identity, benchmark_trace, raw output, validator, rendering, patch report, and review file all identify the remedial run and case-strategic-gated-diligence-001"
      },
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1__rerun1",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1__rerun1.validator.json",
        "validator_sha256": "sha256:a8db52dae255b3bb7a37374fd8f0caee5d398b801e0869a34b86af35cf5257ba",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1__rerun1.patch.json",
        "patch_sha256": "sha256:be452089c33c365a808245e1f81c6df019c8dde64ddb2ba494d248e896e4795f",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__agentic_coding__r1__rerun1.rendering.md",
        "rendering_sha256": "sha256:3321e0edd8899a9585a67547f9d81d7bd25477b60660d8ea58997596d9ea48b7"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun5",
      "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
      "rerun_number": 5,
      "rerun_reason": "Controlled Mode A replacement after four same-shape frontier full-OfOne remedial attempts failed to produce a valid aggregate-eligible benchmark package.",
      "aggregate_policy": "replace_for_aggregate_only",
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "status": "reviewed",
      "execution_mode": "Mode A: Controlled Non-Deep-Research Execution",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun5.md",
      "raw_output_sha256": "sha256:2897a11f82baee33a0161a8d8f64b716017c3fedd9d22a2639c60caa6ad4bf86",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun5.artifact.json",
      "artifact_sha256": "sha256:6bd70e20ba36bf9e63a8bc4c65127a3f1166d30590d1a63df63b2612261a14f0",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun5.md",
      "review_sha256": "sha256:67c79c6fc30535dcb143ae47620b79444186d6ccc7b44087f4c65fc8789f7382",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Controlled Mode A package is case-native, validator-valid, locally reviewed, and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case, prompt, and Mode A contract evidence with stable hashes; missing operating facts remain explicit unknowns",
        "artifact_source_identity": "raw output, artifact JSON, validator JSON, rendering, patch report, and local review all identify the controlled rerun5 slot"
      },
      "benchmark_trace": {
        "case_id": "case-strategic-gated-diligence-001",
        "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun5",
        "case_file": "benchmarks/cases/strategic-gated-diligence.md",
        "case_file_sha256": "sha256:18a0247003e142c80c8748eb4652f900b81ca409b91ba7c370e1362d24680942",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:4168a4e6533f1398611d704254b48c8fbfde5f547a4a0c79cd072a98fdbacd44"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun5.validator.json",
        "validator_sha256": "sha256:de1bbeefa04a96f0da9fb2210290c206e8c421139af5abaa43c2aa6b8850093a",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun5.patch.json",
        "patch_sha256": "sha256:60c289babea552802569575f5a4ff40672835996739841753ff20d3cdcd1ce19",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun5.rendering.md",
        "rendering_sha256": "sha256:116da40a3d99f20b7bce459f9de543e4e4ab0edd87c7ea5c95ee364899d211e5"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1",
      "rerun_of": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1",
      "rerun_number": 1,
      "rerun_reason": "Controlled Mode A replacement for the excluded regulated wastewater frontier full-OfOne repeat-1 artifact after computed local validation failed relation legality and relation-family checks.",
      "aggregate_policy": "replace_for_aggregate_only",
      "case_id": "case-regulated-wastewater-market-entry-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "status": "reviewed",
      "execution_mode": "Mode A: Controlled Non-Deep-Research Execution",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1.md",
      "raw_output_sha256": "sha256:eb8251313f684035e12f5ce9999645c8c1273663549bb2ee0da26021904e7fce",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1.artifact.json",
      "artifact_sha256": "sha256:99869ab95cb00d04ae611efbf89fbe65257c00ab845e02ef0581bb625794a4ee",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1.md",
      "review_sha256": "sha256:929df0ac477a063c2577738673aeb4a606b81c788ccdf3945bcd80b397148c46",
      "aggregate_eligible": true,
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "pass",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": false,
        "reviewer_notes": "Controlled Mode A wastewater package is case-native, validator-valid, locally reviewed, and makes no unsupported superiority claim."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "scenario-level benchmark case, prompt, and Mode A contract evidence with stable case/prompt hashes; missing operating facts remain explicit launch-blocking unknowns",
        "artifact_source_identity": "raw output, artifact JSON, validator JSON, rendering, patch report, and local review all identify the controlled wastewater rerun1 slot"
      },
      "benchmark_trace": {
        "case_id": "case-regulated-wastewater-market-entry-001",
        "run_id": "2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1",
        "case_file": "benchmarks/cases/regulated-wastewater-market-entry.md",
        "case_file_sha256": "sha256:790cd65cf34d0572c9e171127e58aebadbf8ad0c8f09e16d044d6dbbd5b5ec16",
        "prompt_file": "benchmarks/runs/2026-05-17-batch-01/prompts/full_ofone.md",
        "prompt_file_sha256": "sha256:613afac8909b34accb57fd2c24217bb59a28a45860e32f36f6ec5f7f4ab5587e",
        "input_bundle_sha256": "sha256:2849a7b8ab7af3553a448c44916d236e5cdd8dc4c4e37dd9fc5e3a18e4398b0b"
      },
      "machine_artifacts": {
        "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1.validator.json",
        "validator_sha256": "sha256:9af57bff15e7bafc69ebe718f356b68d8e9a85cef11881ffe025fcb5587de793",
        "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1.patch.json",
        "patch_sha256": "sha256:520562c0394148f6fb21be4f4dcec4be93cff9bcfbf31a472f25a199fa1db45f",
        "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-regulated-wastewater-market-entry-001__full_ofone__frontier_reasoning__r1__rerun1.rendering.md",
        "rendering_sha256": "sha256:6992173710a3394fe784f445e65fb2203dfd8fe6b86d17cf15ad6f95e674f5dd"
      }
    }
  ],
  "remedial_attempts": [
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun1",
      "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
      "rerun_number": 1,
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "status": "failed",
      "aggregate_policy": "not_aggregate_eligible",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun1.md",
      "raw_output_sha256": "sha256:c0900989fe10e528648ea57d6f20f1f18f662fcff89bb04194bbc99fb5a9d385",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (36).md",
      "harvested_at": "2026-05-21T00:02:00-06:00",
      "conversation_url": "https://chatgpt.com/c/6a0e8efd-2234-83e8-af43-a7e25266034d",
      "visible_report_metadata": "Research completed in 1h 7m; 10 citations; 117 searches; 20 May; 10 sources; title `Strategic Gated Diligence Remedial Run Research Report`.",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun1.md",
      "review_sha256": "sha256:17250d4766bc7688af9e12e887057e1dcfbd1ac9b7a0a9e57d2d47d3dc8cb86d",
      "failure_reason": "Completed report did not begin with `# Benchmark Raw Output` and did not contain actual `## Artifact JSON`, `## Validator Result`, `## Rendering`, or `## Patch Report` sections.",
      "pre_score_compliance": {
        "case_fidelity": "fail",
        "required_outputs": "fail",
        "independence": "unknown",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Deep Research returned an advisory research report rather than the required benchmark package."
      },
      "semantic_fidelity": {
        "case_binding": "fail",
        "copied_example_risk": "unknown",
        "evidence_provenance_adequacy": "advisory report only; no benchmark artifact, validator result, rendering, or patch report exists",
        "artifact_source_identity": "raw export and review identify the failed remedial run; no artifact object exists"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun2",
      "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
      "rerun_number": 2,
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "status": "failed",
      "aggregate_policy": "not_aggregate_eligible",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun2.md",
      "raw_output_sha256": "sha256:dfdae1034abf0e0521df5103bfa297c605ac5ef149b4a3f85490f070e7179bd8",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (37).md",
      "harvested_at": "2026-05-21T01:08:00-06:00",
      "conversation_url": "https://chatgpt.com/c/6a0ea350-3584-83e8-9d3e-ab7759c489f6",
      "visible_report_metadata": "Research completed in 44m; 1 citation; 20 searches; 21 May; 1 source; title `Benchmark Raw Output`; run metadata shows Status: `completed`.",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun2.artifact.json",
      "artifact_sha256": "sha256:b0673f1297a87f5f9d9f1e36b853ba9d60f847b528632dc0f5e2595953cab809",
      "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun2.validator.json",
      "validator_sha256": "sha256:9243310cef7b30911029959612479caf44bfdcfaa0d246445d22514cac836226",
      "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun2.rendering.md",
      "rendering_sha256": "sha256:197a5cc2952c5bdcb818798963b0f447070de6cc41ac91eeb203eb91acf5869c",
      "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun2.patch.json",
      "patch_sha256": "sha256:514e45e7ae01d74a52430bb19900817348bce920109e149a9f67fcf37df94e7d",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun2.md",
      "review_sha256": "sha256:64934d719e938176e51d9fa4c89a3626e45c1fc3859664ea0f6568152117e08f",
      "failure_reason": "Computed local validation failed because evidence E1, E2, and E3 lack required movement_jobs fields and the tradeoff surface has a reversal-condition defect around G1.",
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "fail",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Benchmark raw-output package shape was present, but computed local validation failed."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "benchmark trace hashes are present, but evidence nodes omit required movement_jobs",
        "artifact_source_identity": "raw output, artifact, validator, rendering, patch report, and review identify the failed remedial rerun2 slot"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun3",
      "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
      "rerun_number": 3,
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "status": "failed",
      "aggregate_policy": "not_aggregate_eligible",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun3.md",
      "raw_output_sha256": "sha256:b64a604e28a5e26d871dc5bca05e4be33dd0630afbbfca8d3d28cbc579e7db85",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (38).md",
      "harvested_at": "2026-05-21T01:58:41-06:00",
      "conversation_url": "https://chatgpt.com/c/6a0eb57b-6b08-83e8-a3e2-16e26adc497f",
      "visible_report_metadata": "Research completed in 19m; 5 citations; 23 searches; 21 May; 5 sources; title `Benchmark Raw Output`; status field shows `completed`.",
      "artifact_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun3.artifact.json",
      "artifact_sha256": "sha256:f3ed71eadec19cfd0cf38b420781d522ee26207fc2dbc9a7997c166f4376ac2a",
      "validator_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun3.validator.json",
      "validator_sha256": "sha256:5f3c252b1ed55915d3b6e81ad81862b2732b822b51752d70aa9d62b7038c7182",
      "rendering_md": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun3.rendering.md",
      "rendering_sha256": "sha256:b47e2c0c4284df2db391887ea15c0a575c4ca6e41d1d01a3e66036e8b85aecb5",
      "patch_json": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun3.patch.json",
      "patch_sha256": "sha256:bd5873d9c3f0bb549f0f09356e73d351cd5202903a9346be4560a1077cb2654e",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun3.md",
      "review_sha256": "sha256:ae21e3aa65ec56903de43279c6cc865a6ea95d8f33b181c032ec5d6cca7f7245",
      "failure_reason": "Computed local validation failed because benchmark_trace lacks required current-schema fields and edges X2, X3, and X4 use illegal endpoint/relation combinations; raw export also omitted exact top-level run metadata required by the packet.",
      "pre_score_compliance": {
        "case_fidelity": "pass",
        "required_outputs": "fail",
        "independence": "pass",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Benchmark package sections were present, but required run metadata and executable validation did not pass."
      },
      "semantic_fidelity": {
        "case_binding": "pass",
        "copied_example_risk": "low",
        "evidence_provenance_adequacy": "case-bound evidence includes movement_jobs, but benchmark_trace shape and graph relations fail current validation",
        "artifact_source_identity": "raw output, artifact, validator, rendering, patch report, and review identify the failed remedial rerun3 slot"
      }
    },
    {
      "run_id": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun4",
      "rerun_of": "2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1",
      "rerun_number": 4,
      "case_id": "case-strategic-gated-diligence-001",
      "arm_id": "full_ofone",
      "model_family": "frontier_reasoning",
      "repeat": 1,
      "status": "failed",
      "aggregate_policy": "not_aggregate_eligible",
      "raw_output": "benchmarks/runs/2026-05-17-batch-01/outputs/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun4.md",
      "raw_output_sha256": "sha256:5d4de650a1c2f3f6612b718f45e7313f40433dfaf851afeaad1b3ca0f9dbd702",
      "source_download": "/Users/jamesbrady/Downloads/deep-research-report (39).md",
      "harvested_at": "2026-05-21T02:44:16-06:00",
      "conversation_url": "https://chatgpt.com/c/6a0ec0a2-3814-83e8-8f86-23b625eace67",
      "visible_report_metadata": "Research completed in 18m; 12 citations; 28 searches; 21 May; 12 sources; title `Running an Unspecified OfOne Benchmark Packet Exactly`.",
      "review_file": "benchmarks/reviews/2026-05-17-batch-01/2026-05-17-batch-01__case-strategic-gated-diligence-001__full_ofone__frontier_reasoning__r1__rerun4.md",
      "review_sha256": "sha256:af4a243df6834d66ebd10b45e7c746f103c19177493cce860ce869fd29a347ca",
      "failure_reason": "Completed report was a meta/advisory analysis of an unspecified packet and did not begin with `# Benchmark Raw Output`, include exact run metadata, or contain actual case-bound `## Artifact JSON`, `## Validator Result`, `## Rendering`, and `## Patch Report` sections.",
      "pre_score_compliance": {
        "case_fidelity": "fail",
        "required_outputs": "fail",
        "independence": "unknown",
        "no_superiority_compliance": "pass",
        "auto_reject": true,
        "reject_reason": "Deep Research returned a meta/advisory report rather than the required benchmark package."
      },
      "semantic_fidelity": {
        "case_binding": "fail",
        "copied_example_risk": "unknown",
        "evidence_provenance_adequacy": "advisory report only; no benchmark artifact, validator result, rendering, or patch report exists",
        "artifact_source_identity": "raw export and review identify the failed remedial rerun4 slot; no artifact object exists"
      }
    }
  ]
}
