{
  "schema_version": 1,
  "report_version": "draft1.0",
  "date_utc": "2026-10-06",
  "claims": [
    {
      "id": "C01",
      "claim": "120 pilot tasks, four candidates and four judge calls per task, with 960 accounted requests.",
      "evidence": [
        "RESULTS.json:collection_usage.requests",
        "RESULTS.json:candidate_passes.hidden_total",
        "collector plan:candidates_per_task"
      ],
      "status": "measured and reconciled"
    },
    {
      "id": "C02",
      "claim": "First candidate: 109/120; higher-cap SoftBoN: 329/360 seed decisions, paired effect +0.00555556 with CI [0, 0.01666667].",
      "evidence": [
        "RESULTS.json:results",
        "completed analysis commitment",
        "independent metadata audit"
      ],
      "status": "measured descriptive;no superiority claim"
    },
    {
      "id": "C03",
      "claim": "Fixed 49-task subset, 196 public slots, all policies 47/49; single remains an unfiltered comparator.",
      "evidence": [
        "RESULTS.json:grading_slots",
        "RESULTS.json:results",
        "RESULTS.json:single_comparator",
        "frozen replay source"
      ],
      "status": "measured and source checked"
    },
    {
      "id": "C04",
      "claim": "437/480 passing candidates; 108 all-pass, ten all-fail, two mixed pools; oracle 110/120 and headroom 1/120.",
      "evidence": [
        "RESULTS.json:posthoc_candidate_pool_diagnostic",
        "independent boolean-oracle metadata audit"
      ],
      "status": "post hoc;not primary endpoint"
    },
    {
      "id": "C05",
      "claim": "276,882 prompt plus 100,216 returned tokens, 377,098 total; external paid spend $0.",
      "evidence": [
        "RESULTS.json:collection_usage"
      ],
      "status": "observed APIusage;electricityhardwarelaborunmeasured"
    },
    {
      "id": "C06",
      "claim": "Four unique source-contract failures outside the public subset; zero judge abstentions.",
      "evidence": [
        "RESULTS.json:generation_contract_failures",
        "RESULTS.json:collection_usage.judge_abstentions",
        "case manifest automatic outcomes audited4hidden0public"
      ],
      "status": "retained failures;cause not inferred"
    },
    {
      "id": "C07",
      "claim": "120 hidden references pass; original 49/57 public eligibility retained after new qualification passes 51/57.",
      "evidence": [
        "RESULTS.json:prospective_controls",
        "precollection gate commitment",
        "prospective case-manifest qualification"
      ],
      "status": "qualified;historicalcasebyteparityunavailable"
    },
    {
      "id": "C08",
      "claim": "Temperature 0.2, context 4,096, output caps 512/128, think=false, seed 20261006+sample, no retries.",
      "evidence": [
        "launch source commitment",
        "collector plan/source",
        "collector request construction"
      ],
      "status": "declared executed source settings;no fullenvironmentlock"
    },
    {
      "id": "C09",
      "claim": "Caps 4,608/9,216/18,432/36,864, three cached replay seeds; worst-case generation/judge reservations 4,608/4,224.",
      "evidence": [
        "RESULTS.json:results",
        "frozen token_replay_v5.py",
        "frozen replay.py defaults"
      ],
      "status": "counterfactual cached replay;not actualcollectioncost"
    },
    {
      "id": "C10",
      "claim": "9,288 decisions frozen before hidden analysis; 40 groups with independently recomputed task bootstrap.",
      "evidence": [
        "RESULTS.json:decision_count",
        "frozen decisions commitment",
        "same-owner independent-agent audit commitment"
      ],
      "status": "ownerreported chronology/internal hashes;not trustedoutsidecustody"
    },
    {
      "id": "C11",
      "claim": "Five owned networkless candidate batches, at most 180 cases each; 256 MiB memory, 16 tasks, CPU 3 seconds, inner wall 5 and outer wall 15 seconds.",
      "evidence": [
        "candidate reportlimits checked",
        "case manifest batch count",
        "qualified host/guest source and SHA closure"
      ],
      "status": "engineering qualification;CPUusageunmeasured;maliciousgraderintegritynotproven"
    },
    {
      "id": "C12",
      "claim": "HumanEval 164 records deterministically assigned 120/20/20/4; upstream revision, retrieval and MIT license pinned.",
      "evidence": [
        "data/source.json",
        "freeze_tasks.py",
        "task manifest/dataset commitments"
      ],
      "status": "source checked;not model-pretraining nonexposure"
    },
    {
      "id": "C13",
      "claim": "Earlier development retained fourteen formatting failures; separate structured diagnostic had twelve passing candidates.",
      "evidence": [
        "canonical historicaldevelopment section"
      ],
      "status": "historical engineering evidence;not pooled"
    },
    {
      "id": "C14",
      "claim": "Ollama 0.35.1 serving installation identified by the execution owner.",
      "evidence": [
        "executionowner pre-runinventory and canonical report"
      ],
      "status": "ownerreported servingversion;fullbinary/environmentlocknotprovided"
    },
    {
      "id": "C15",
      "claim": "Existing imperfect-verifier and BoN/SoftBoN/Poisson work is contextual attribution, not a novelty or theorem-transfer claim.",
      "evidence": [
        "arxiv2411.17501v3primarypagechecked20261006",
        "arxiv2506.19248v2primarypagechecked20261006",
        "arxiv2107.03374primarypagechecked20261006"
      ],
      "status": "context only;no theoremtransfer/exactauthor-systemreproduction"
    }
  ],
  "not_claimed": [
    "Completed H1/H2/H3 studies",
    "Novel method or model architecture",
    "Confirmatory superiority or equivalence",
    "Frontier transfer or training",
    "Independent outside replication or peer review",
    "Measured CPU usage, FLOPs or GPU time",
    "Zero total economic cost",
    "Fresh benchmark or absence of contamination",
    "Public runnable full-source release"
  ],
  "source_commit": "ab5cd6b56c2d38059e09687ffc12b8cec6778d4e",
  "launch_metadata_sha256": "8ce6ea11c2609fe8fc3060be0a438fc48c435b54eea70fbcb4ff48b572fa0d14",
  "source_lock": [
    {
      "path": "examples/verifier-reliability/data/pilot_collection_plan_v5.json",
      "sha256": "db6222052aa8f97d07d5eab12df837bf908ba08f4982af06f10ff01d32a3e1ff"
    },
    {
      "path": "examples/verifier-reliability/docs/PILOT-PROTOCOL-V5-20261006.md",
      "sha256": "7b9587955e1985a3bde3379fbd91202e90eb5b18e3b32276ab9e4ff254ca4ed2"
    },
    {
      "path": "examples/verifier-reliability/local_collect_pilot_v5.py",
      "sha256": "5dac8edbb4070c79a3f4340e644f441ef4ffa3919771128ad4e0de3be4c3aec5"
    },
    {
      "path": "examples/verifier-reliability/tests/test_collector_pilot_v5.py",
      "sha256": "8e30235cd04274ae302f6e545fc2ba83ea4917ab263be3944cc6590ffeb8bded"
    },
    {
      "path": "examples/verifier-reliability/tests/test_failures_v5.py",
      "sha256": "4de238d1917e999489d7c548e7f27a97bd669c1f228f24f2121ac5c9bacafce5"
    },
    {
      "path": "examples/verifier-reliability/tests/test_token_replay_v5.py",
      "sha256": "8cbe931e35dc5cbd3111d79397f20e706b426b92f409d7357e1bd4544a4d32fc"
    },
    {
      "path": "examples/verifier-reliability/token_replay_v5.py",
      "sha256": "5142d7a9ad53c227a747f5c3743878a720463d5d63422516348ed5dd3f6badc8"
    }
  ],
  "aggregate_sha256": "3ea3b1a6fe2daf407fbb7fd1bf98e329cee69eb03c983534bb3c295cf1404e0e",
  "reviewed_artifact_commitments": {
    "analysis": "4fcd3b98255aa15d7a6f58a303a159fb143f9a411c4bc4426c4d424bc67df3cb",
    "frozen_decisions": "da9a0cc203fbc44e7511a384cf2257f11becc18c06ab72d8735e44bfdf4d4d25",
    "grading_coverage": "a1633eba0b239bd81c4b984a56de3267f60d0f330c0e325d957895fec9e330c2",
    "case_manifest": "d5b3420c07d57099c630bdd457e55c614ee13ad03d97eeec02f4bb388ab98c9b",
    "independent_same_owner_metadata_audit": "3516212a250e466ae0dd5819a7a44f5d4d545559c1a79fe2fb81cabc7c24a445"
  },
  "private_content_excluded": true,
  "source_contribution": "Original integration, accounting, qualified execution, replay and reconciliation; upstream models, data and established methods are credited."
}
