{
  "schema_version": "bonfire.research-evidence.v1",
  "generated_at": "2026-07-29",
  "description": "Machine-readable provenance for the headline experiment claims displayed on bonfiredb.dev. This is a curated public index, not a replacement for the canonical experiment ledger.",
  "claims": [
    {
      "id": "e1-three-arm-secondary-comparison",
      "status": "measured",
      "population": "409 MIMIC-IV-on-FHIR benchmark questions; overflow stratum n=262",
      "sample_size": 409,
      "inference_unit": "question",
      "preregistration": "Historical secondary comparison; not treated as a Bonfire confirmatory result.",
      "artifact_availability": "Aggregate report is public. Some historical raw-provider and cost receipts were not retained.",
      "claim_boundary": "Describes a secondary comparison on one benchmark and model setup. It does not validate Bonfire, graph storage, or a universal raw-FHIR ceiling. The arms did not share one server instance, the strata are post-hoc, and the historical multi-turn A0′ (39.4%) is not comparable to the frozen-packet A0′ baseline (44.7%) used in A6a.",
      "displayed_values": ["25.4%", "39.4%", "65.3%", "0/262"],
      "site_occurrences": [
        {
          "path": "src/lib/data/experiments.ts",
          "must_contain": ["25.4%", "39.4%", "65.3%"]
        },
        {
          "path": "src/fragments/problems.html",
          "must_contain": ["0/262"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "af93eb968fc33b694a0da7b58b46ad27f928c5eb",
        "path": "docs/FINAL_REPORT.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/af93eb968fc33b694a0da7b58b46ad27f928c5eb/docs/FINAL_REPORT.md",
        "sha256": "07a748a88cf1075990c0b8d187dd08589b77d8f2b5771256d7bc69a1f39af691"
      }
    },
    {
      "id": "a6a-initial-selection",
      "status": "measured",
      "population": "409 paired MIMIC-IV-on-FHIR questions across 90 patient clusters",
      "sample_size": 409,
      "inference_unit": "paired question, with patient-cluster bootstrap uncertainty",
      "preregistration": "Pre-registered confirmatory comparison; frozen selector and baseline.",
      "artifact_availability": "Public result report, preregistration, grading result, and packet artifacts are retained in the eval repository.",
      "claim_boundary": "Licenses only that deterministic question-only selection beat the frozen query-blind projection in this paired run. Both prompts omitted the same time assumption, so absolute accuracies are lower bounds. The −43% is packet payload characters (36.7M vs 64.0M), not tokens; token receipts for this run were not retained.",
      "displayed_values": ["54.3%", "44.7%", "+9.5pp", "−43%"],
      "site_occurrences": [
        {
          "path": "src/lib/data/experiments.ts",
          "must_contain": ["54.3%", "44.7%", "+9.5pp", "−43%"]
        },
        {
          "path": "src/pages/index.astro",
          "must_contain": ["54.3%", "44.7%", "+9.5pp"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "af93eb968fc33b694a0da7b58b46ad27f928c5eb",
        "path": "docs/A6A_RESULT.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/af93eb968fc33b694a0da7b58b46ad27f928c5eb/docs/A6A_RESULT.md",
        "sha256": "f87aa81e427bda549edc2d7fec3e1dc506cd0fc175e4ab752a4ceda406818c65"
      }
    },
    {
      "id": "a6a-repaired-replication",
      "status": "measured",
      "population": "409 paired MIMIC-IV-on-FHIR questions across 90 patient clusters",
      "sample_size": 409,
      "inference_unit": "paired question, with patient-cluster bootstrap uncertainty",
      "preregistration": "Assumption-fixed replication with a pinned model runtime.",
      "artifact_availability": "Aggregate result and token economics are consolidated in the public canonical evidence ledger.",
      "claim_boundary": "Replicates the narrow selection-over-projection result after repairing the shared time assumption. It does not establish cross-model or cross-server generality. The −35.5% is accepted-attempt model tokens, a different unit from the initial run's payload-character figure.",
      "displayed_values": ["54.5%", "46.2%", "+8.3pp", "−35.5%"],
      "site_occurrences": [
        {
          "path": "src/lib/data/experiments.ts",
          "must_contain": ["54.5%", "46.2%", "+8.3pp", "−35.5%"]
        },
        {
          "path": "src/pages/index.astro",
          "must_contain": ["54.5%", "46.2%", "+8.3pp"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "af93eb968fc33b694a0da7b58b46ad27f928c5eb",
        "path": "docs/results/EXPERIMENT_EVIDENCE_LEDGER.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/af93eb968fc33b694a0da7b58b46ad27f928c5eb/docs/results/EXPERIMENT_EVIDENCE_LEDGER.md",
        "sha256": "d90803c69e96440114e0880f69e797bc5442ec277001d9ad83a70fae04056cc9"
      }
    },
    {
      "id": "qt4-fixed-microbiology-vocabulary",
      "status": "measured",
      "population": "374-question untouched valid-split holdout: 44 registered microbiology questions and 330 byte-identical controls",
      "sample_size": 374,
      "inference_unit": "paired question; primary treatment inference on the registered 44-question stratum",
      "preregistration": "Pre-registered confirmatory holdout with a promotion rule and safety bound.",
      "artifact_availability": "Public aggregate result and forensic audit are retained. Healthcare-derived answer content is not published.",
      "claim_boundary": "Promotes fixed microbiology vocabulary only. The incremental traversal contrast was unresolved and was not promoted.",
      "displayed_values": ["10/44 · 22.7%", "25/44 · 56.8%", "+34.1pp", "p=0.000275", "29/44"],
      "site_occurrences": [
        {
          "path": "src/pages/research/experiments/vocabulary.astro",
          "must_contain": ["10/44 · 22.7%", "25/44 · 56.8%", "+34.1pp", "p=0.000275", "29/44"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "af93eb968fc33b694a0da7b58b46ad27f928c5eb",
        "path": "docs/results/QT4_VALID374_RESULT.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/af93eb968fc33b694a0da7b58b46ad27f928c5eb/docs/results/QT4_VALID374_RESULT.md",
        "sha256": "137116d6646d86d897c5aa03180cabc6034ac5f0b0843856f9378acc58db30a8"
      }
    },
    {
      "id": "qt1-3-packet-structure-nulls",
      "status": "null",
      "population": "Three paired 409-question MIMIC-IV-on-FHIR comparisons",
      "sample_size": 409,
      "inference_unit": "paired question, with patient-cluster bootstrap uncertainty",
      "preregistration": "Query-time feature screens recorded in the ledger as null and not promoted; the ledger does not record a prospective registration for these screens.",
      "artifact_availability": "Aggregate results and accepted-token receipts are consolidated in the public evidence ledger.",
      "claim_boundary": "Pinned references, aggregate summaries, and endpoint reserves did not improve correctness in these tests. The A6a comparator was re-run per screen (53.8–54.8%), so baselines differ slightly across the three tests. The intervals do not prove exact equivalence or rule out effects on a different task.",
      "displayed_values": ["QT-1 +0.24pp", "QT-2 −0.24pp", "QT-3 0.0pp"],
      "site_occurrences": [
        {
          "path": "src/pages/research/experiments/index.astro",
          "must_contain": ["QT-1 +0.24pp", "QT-2 −0.24pp", "QT-3 0.0pp"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "af93eb968fc33b694a0da7b58b46ad27f928c5eb",
        "path": "docs/results/EXPERIMENT_EVIDENCE_LEDGER.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/af93eb968fc33b694a0da7b58b46ad27f928c5eb/docs/results/EXPERIMENT_EVIDENCE_LEDGER.md",
        "sha256": "d90803c69e96440114e0880f69e797bc5442ec277001d9ad83a70fae04056cc9"
      }
    },
    {
      "id": "a11-bounded-traversal",
      "status": "bounded-support",
      "population": "120 constructed path-required questions; 96 answerable cases",
      "sample_size": 120,
      "inference_unit": "question, with patient-cluster bootstrap uncertainty",
      "preregistration": "Sealed V/T/E mechanism test with registered primary and secondary contrasts.",
      "artifact_availability": "Public aggregate result and receipts are retained; answer content is not exposed in the site manifest.",
      "claim_boundary": "Shows terminal-evidence recovery on a deliberately constructed path-required corpus. The vocabulary-only baseline abstained on every question, so its 24/120 correct answers are all correct abstentions and its terminal-evidence recall was 0/96. It does not show that graph storage is superior or that event grouping adds value beyond traversal.",
      "displayed_values": ["119/120", "24/120", "96 of 96", "120/120", "0/96"],
      "site_occurrences": [
        {
          "path": "src/fragments/feature-mcp.html",
          "must_contain": ["119/120", "24/120", "96 of 96", "120/120"]
        },
        {
          "path": "src/pages/index.astro",
          "must_contain": ["119/120", "24/120"]
        },
        {
          "path": "src/fragments/research-agent-context-fhir.html",
          "must_contain": ["119/120", "24/120"]
        },
        {
          "path": "src/pages/research/experiments/index.astro",
          "must_contain": ["0/96"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "af93eb968fc33b694a0da7b58b46ad27f928c5eb",
        "path": "docs/results/A11_RESULT.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/af93eb968fc33b694a0da7b58b46ad27f928c5eb/docs/results/A11_RESULT.md",
        "sha256": "8f5b4e389bbe3ce530aa1028c311a799077ac4eab2cfc53cef6a771be19dcedc"
      }
    },
    {
      "id": "a11b-causal-isolation-and-successor",
      "status": "exploratory",
      "population": "384 spent efficacy patients in A11b r3; 64 fresh development patients built for the successor while 384 fresh efficacy patients remain unopened",
      "sample_size": 384,
      "inference_unit": "one question per synthetic patient, paired across three evidence-presentation arms",
      "preregistration": "A11b r3 was sealed, but its behavioral interpretation was superseded after a forensic amendment found that the registered normalization erased insufficiency. The successor requires a new amended seal before any call.",
      "artifact_availability": "The strict r3 artifact, forensic amendment, successor zero-model build, and adaptation gate are public. The first successor seal made zero calls and is retired; the r3 artifact is not runnable as sealed against the current backend.",
      "claim_boundary": "A11b did not establish an event-grouping benefit or a valid behavioral null. A conservative post-hoc sensitivity read reached 314/384 vs 384/384 vs 384/384, which is why the strict tie is not treated as behavioral. The only licensed conclusion is that the task and answer contract required redesign; no feature was promoted.",
      "displayed_values": ["288/384", "26/96", "96/96", "successor reseal required"],
      "site_occurrences": [
        {
          "path": "src/pages/research/experiments/index.astro",
          "must_contain": ["288/384", "26/96", "96/96", "successor reseal required"]
        },
        {
          "path": "src/fragments/research-agent-context-fhir.html",
          "must_contain": ["288/384"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "07a264412066f27993797ed072a23f2435b54dc2",
        "path": "docs/results/EXPERIMENT_EVIDENCE_LEDGER.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/07a264412066f27993797ed072a23f2435b54dc2/docs/results/EXPERIMENT_EVIDENCE_LEDGER.md",
        "sha256": "45a9b47a8a126f79aff3f35ce064c02787c34fa797c100e878b884884a5e76e4"
      }
    },
    {
      "id": "w1a-deterministic-prejoin",
      "status": "exploratory",
      "population": "409 paired questions across 90 patient clusters; registered visit-specific subset n=176",
      "sample_size": 409,
      "inference_unit": "paired question, with patient-cluster bootstrap uncertainty",
      "preregistration": "Local protocol predated the run but lacked an independent pre-run Git anchor, was modified after results existed, and did not retain the answer model or reasoning-effort pin.",
      "artifact_availability": "Public aggregate result is retained; opaque-ID grading sensitivity remains pending.",
      "claim_boundary": "Supports carrying deterministic prejoin forward as an exploratory mechanism. The +6.8-point subset estimate is not a pooled result and is not yet grading-robust.",
      "displayed_values": ["56.0%", "54.0%", "+2.0pp", "68.2%", "61.4%", "+6.8pp", "−1.8pp"],
      "site_occurrences": [
        {
          "path": "src/pages/research/experiments/prejoin-test.astro",
          "must_contain": ["56.0%", "54.0%", "+2.0pp", "68.2%", "61.4%", "+6.8pp", "−1.8pp"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "f4d276acd76a5bf27caa46be8e6c30706d26367f",
        "path": "docs/results/W1A_RESULT.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/f4d276acd76a5bf27caa46be8e6c30706d26367f/docs/results/W1A_RESULT.md",
        "sha256": "e78a021ac54335e530e84cd167784fc797a42cbe07a58ad16d18a42753e95731"
      }
    },
    {
      "id": "w1a-payload-economics",
      "status": "exploratory",
      "population": "409 paired packet builds; join-triggered subset n=176",
      "sample_size": 409,
      "inference_unit": "packet payload characters",
      "preregistration": "Reported in the preserved local protocol as a descriptive metric, not a gate.",
      "artifact_availability": "The payload figures appear in the preserved local protocol document; per-packet payload receipts are not separately published.",
      "claim_boundary": "Payload characters, not tokens or dollar cost: 26.7M vs 41.4M characters across all 409 packets (−36%) and a 20k vs 35k median on the 176 join-triggered questions (−43%). W1A's token receipts (22.1M vs 28.8M cumulative input) are a separate, smaller delta.",
      "displayed_values": ["−36% · −43%"],
      "site_occurrences": [
        {
          "path": "src/pages/research/experiments/prejoin-test.astro",
          "must_contain": ["−36% · −43%"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "f4d276acd76a5bf27caa46be8e6c30706d26367f",
        "path": "docs/prereg/W1A_LOCAL_PROTOCOL_PRESERVATION.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/f4d276acd76a5bf27caa46be8e6c30706d26367f/docs/prereg/W1A_LOCAL_PROTOCOL_PRESERVATION.md",
        "sha256": "6a376095423144cdef4e55cf438e672d1fd617ef4f3be3478ccb1f2224b118f6"
      }
    },
    {
      "id": "w2a-agent-side-join",
      "status": "exploratory",
      "population": "176 paired visit-specific questions",
      "sample_size": 176,
      "inference_unit": "paired question, with patient-cluster bootstrap uncertainty",
      "preregistration": "Local protocol predated the run but lacked an independent pre-run Git anchor; the historical panel exposed arm labels and did not retain a complete answer-model pin.",
      "artifact_availability": "Public aggregate result is retained; opaque-ID grading sensitivity remains pending.",
      "claim_boundary": "No accuracy difference was established and equivalence was not tested. Only the input-token gap and discordance pattern are descriptive findings. The 191,721 and 47,205 figures are per-question means (run totals divided by 176).",
      "displayed_values": ["72.2%", "68.2%", "+4.0 points", "4.061×", "53/176", "191,721", "47,205"],
      "site_occurrences": [
        {
          "path": "src/pages/research/experiments/agent-join.astro",
          "must_contain": ["72.2%", "68.2%", "+4.0 points", "4.061×", "53/176", "191,721", "47,205"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "f4d276acd76a5bf27caa46be8e6c30706d26367f",
        "path": "docs/results/W2A_RESULT.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/f4d276acd76a5bf27caa46be8e6c30706d26367f/docs/results/W2A_RESULT.md",
        "sha256": "ce292fd5c7715c5b8f1d45eeb2344dee92e09c90342fe1ec4f12ebc16eebfed2"
      }
    },
    {
      "id": "judge-sensitivity-regrade",
      "status": "measured",
      "population": "111 numeric arm answers in the initial judge validation; later three-vote sensitivity panels",
      "sample_size": 111,
      "inference_unit": "answer-grade decision",
      "preregistration": "Forensic re-grade rather than a prospectively registered efficacy comparison.",
      "artifact_availability": "Public aggregate forensic report is retained. The underlying runs/ judge-leaderboard files are local-only (gitignored); only the in-report tables are public.",
      "claim_boundary": "Demonstrates that the initial single-judge setup was unreliable and that later panels were more accurate under a different prompt/context. The zero-false-positive figure is the gpt-5-mini setup; a later panel recorded one false positive. The benchmark's shipped default judge was not measured, so this is not a fair model leaderboard.",
      "displayed_values": ["61.3%", "98–99%", "43 false negatives", "0 false positives"],
      "site_occurrences": [
        {
          "path": "src/lib/data/experiments.ts",
          "must_contain": ["61.3%", "98–99%", "43 false negatives", "0 false positives"]
        }
      ],
      "source": {
        "repository": "https://github.com/coralehr/fhir-mcp-eval",
        "commit": "af93eb968fc33b694a0da7b58b46ad27f928c5eb",
        "path": "docs/TRUSTWORTHY_REGRADE.md",
        "url": "https://github.com/coralehr/fhir-mcp-eval/blob/af93eb968fc33b694a0da7b58b46ad27f928c5eb/docs/TRUSTWORTHY_REGRADE.md",
        "sha256": "0441cae38c06bf13db25ec6b61c79ad4f7be727700c47897f1d74a6b50da1f0b"
      }
    }
  ]
}
