{
  "contract": "this hash is committed to git BEFORE the benchmark runs; a matching observed outcome then proves the prediction was pre-registered, not fit post-hoc.",
  "name": "protocol_bench_qwen_bar",
  "note": "S06 candidate 4. Sealed before any prompt was issued and before the weights existed on this host. protocol-bench has never scored a model in this repository; the labels predate the run by construction.",
  "prediction": {
    "bar": {
      "FAIL_publishable_as": "the measured numbers with the prediction that missed named explicitly; a small model scoring above 0.70 or replaying a real counterexample is a MORE interesting result than one scoring inside the band, and is published as such.",
      "PASS": "all three predictions hold on the first and only scoring run.",
      "no_second_run": "the completions are scored once; no re-prompting, no decoding changes."
    },
    "benchmark": {
      "ground_truth": "oss/protocol-bench/src/protocol_bench/data/ground_truth.json",
      "labels": {
        "CANDIDATE_COUNTEREXAMPLE": 1,
        "KNOWN_COUNTEREXAMPLE": 1,
        "PROVEN_SAFE": 13
      },
      "n_tasks": 15,
      "name": "protocol-bench",
      "why_prereg_by_construction": "the labels ship inside the wheel and on the public dataset card and predate every completion; the model cannot have been fit to them by this lane because this lane has never scored any model against them."
    },
    "both_outcomes_publishable": true,
    "candidate": "S06-4 protocol-bench scored against a bar sealed before the first prompt",
    "kill_criterion": "the weights cannot be fetched, or mlx_lm cannot generate on this host: record NOT_RUN with the blocker, no retry budget.",
    "model": {
      "decoding": "greedy, temperature 0, one completion per task, max_tokens 512",
      "note": "no weights are cached on this host at sealing time (11 MB of tokenizer files only); the fetch is part of the run and its failure is the kill criterion.",
      "open_weights": true,
      "paid_api": false,
      "repo": "mlx-community/Qwen2.5-1.5B-Instruct-4bit",
      "runtime": "mlx_lm 0.31.2"
    },
    "negative_control_run_first": "a fabricated-completions file -- plausible prose with a non-replaying trace on every task -- must score ZERO valid counterexamples. If fabrication scores above zero the scorer is unsound and no model result is reported.",
    "predictions": {
      "balanced_accuracy_in": [
        0.4,
        0.7
      ],
      "krack_trace_replays": false,
      "rationale": "a 1.5B instruction model asked to emit a machine-checkable counterexample trace for a published protocol defect is expected to answer plausibly and replay badly; the interesting quantity is the gap between stated accuracy and replayed accuracy.",
      "valid_replayed_counterexamples_at_most": 1
    }
  },
  "schema": "prereg-seal-v1",
  "sealed_at_commit": "1d5b14cbdedbad31eca050c71613f609396a83c3",
  "sha256": "640df291c2e37bf02ff090524d752e4b42773041ba79d37402fdff89f8027d50"
}
