{
  "$schema": "../schema/experiment-card-v0.md",
  "registry_version": "0.1",
  "updated": "2026-05-25",
  "totals": {
    "experiments": 6,
    "pending": 1,
    "runs": 4337,
    "domains": 4,
    "license": "CC BY 4.0"
  },
  "experiments": [
    {
      "id": "EXP-006",
      "slug": "exp-006",
      "title": "Confirmatory E1+E2 — outcomes-only insufficiency",
      "type": "confirmatory",
      "status": "confirmed",
      "date": "2026-05-25",
      "paper": "Paper 2 · Delegation-Grade Agents · v1.1",
      "domain": "refund, claim, memory, code",
      "evidence": "540 runs · 4 domains · 1 model · 1 runtime",
      "protocol": "PAPER2_PROTOCOL_FREEZE.json",
      "corpus": "OPBR-Bench v0",
      "result_num": "E1 + E2",
      "result_text": "Outcomes-only insufficiency replicated under preregistration. Behavior contracts dominate output-only and rich-output baselines on F1 under canonical OPBR.",
      "url": "/v3/lab/exp/exp-006.html"
    },
    {
      "id": "EXP-004",
      "slug": "exp-004",
      "title": "Held-out repair — contracts vs. baselines on canonical OPBR",
      "type": "experiment",
      "status": "confirmed",
      "date": "2026-05-10",
      "paper": "Paper 1 · Study 5",
      "domain": "refund, claim, memory, code",
      "evidence": "400 runs · 4 domains",
      "protocol": "PAPER1_PROTOCOL_FREEZE.json#study-5",
      "corpus": "OPBR-Bench v0 (held-out partition)",
      "result_num": "F1 = 0.982",
      "result_text": "Behavior contracts F1 = 0.982 (precision 0.964, recall 1.000) on held-out OPBR. Best baseline (rich-output provenance) F1 = 0.400. ΔF1 = 0.582. McNemar paired vs every baseline: p ≈ 0.",
      "url": "/v3/lab/exp/exp-004.html"
    },
    {
      "id": "EXP-002",
      "slug": "exp-002",
      "title": "Semi-blind perturbation — 4-domain field rate",
      "type": "experiment",
      "status": "confirmed",
      "date": "2026-04-28",
      "paper": "Paper 1 · Study 2",
      "domain": "refund, claim, memory, code",
      "evidence": "1,437 runs · 4 domains · 6 conditions",
      "protocol": "PAPER1_PROTOCOL_FREEZE.json#study-2",
      "corpus": "OPBR-Bench v0",
      "result_num": "θ = 0.914",
      "result_text": "Under non-instructed perturbations (latency, prompt compression, tool-semantics ambiguity, evidence noise), 91.4% of behavioral regressions produce passing outputs. Wilson 95% CI [0.894, 0.932].",
      "url": "/v3/lab/exp/exp-002.html"
    },
    {
      "id": "EXP-003",
      "slug": "exp-003",
      "title": "Action-affordance merge — proposal/commit ambiguity",
      "type": "experiment",
      "status": "confirmed",
      "date": "2026-04-22",
      "paper": "Paper 1 · Study 4",
      "domain": "refund, claim, memory, code",
      "evidence": "960 runs · 4 domains",
      "protocol": "PAPER1_PROTOCOL_FREEZE.json#study-4",
      "corpus": "OPBR-Bench v0",
      "result_num": "all detectors degrade",
      "result_text": "Merging propose and commit into a single ambiguous tool degrades every structural detector. Action ontology is part of the contract, not orthogonal to it.",
      "url": "/v3/lab/exp/exp-003.html"
    },
    {
      "id": "EXP-001",
      "slug": "exp-001",
      "title": "Controlled mechanism — induced fast-lane vs. baseline",
      "type": "experiment",
      "status": "confirmed",
      "date": "2026-04-12",
      "paper": "Paper 1 · Study 1",
      "domain": "refund, claim, memory",
      "evidence": "1,080 runs · 3 domains · 6 conditions",
      "protocol": "PAPER1_PROTOCOL_FREEZE.json#study-1",
      "corpus": "OPBR-Bench v0",
      "result_num": "180 / 180",
      "result_text": "180/180 induced fast-lane runs pass output evaluation while behavior contracts fail. Drift paths also pass output+operational composite (60/60). Only the behavior layer separates them.",
      "url": "/v3/lab/exp/exp-001.html"
    },
    {
      "id": "EXP-005",
      "slug": "exp-005",
      "title": "Completeness analyzer — Study 1 contract specification audit",
      "type": "heuristic",
      "status": "heuristic",
      "date": "2026-04-15",
      "paper": "Paper 1 · Study 7",
      "domain": "refund, claim, memory",
      "evidence": "Static analysis of Study 1 contracts vs. empirical traces",
      "protocol": "PAPER1_PROTOCOL_FREEZE.json#study-7",
      "corpus": "Study 1 trace corpus (1,080 runs)",
      "result_num": "no missing actions",
      "result_text": "The completeness analyzer finds no load-bearing missing consequential action in Study 1 contracts. The Study 2 recall gap is enumerable specification work, not a flaw in the mechanism.",
      "url": "/v3/lab/exp/exp-005.html"
    },
    {
      "id": "EXP-007",
      "slug": "exp-007",
      "title": "Cross-runtime — Codex CLI replication",
      "type": "experiment",
      "status": "preregistered",
      "date": "2026-06 (planned)",
      "paper": "Paper 1 · Study 6 (Gate E)",
      "domain": "code, refund",
      "evidence": "n preregistered; runs pending Codex install + API key",
      "protocol": "PAPER1_PROTOCOL_FREEZE.json#study-6",
      "corpus": "OPBR-Bench v0 (cross-runtime partition)",
      "result_num": "PENDING",
      "result_text": "Replication of Study 5 contract-vs-baseline comparison under Codex CLI runtime, with Claude Code as cross-runtime reference. Tests whether contract dominance generalizes across runtime.",
      "url": "/v3/lab/exp/exp-007.html"
    }
  ]
}
