{
  "frozen_at": "2026-07-18T23:46:57.951556+00:00",
  "git_sha": "f93f4b5578c221d3d2665f9b305f3e86759be862+dirty",
  "spec": {
    "analysis": "campaign.err.analyze",
    "hypotheses": [
      {
        "id": "H-err-qwen",
        "prediction": {
          "ci_excludes": 0.0,
          "comparator": ">",
          "effect": null,
          "metric": "err_auroc_delta_qwen8b",
          "threshold": 0.03
        },
        "scoreboard_row": null,
        "statement": "internal features beat the exposed seen-pair margin at predicting held-out RB2 completion errors on Qwen3-8B"
      },
      {
        "id": "H-err-llama",
        "prediction": {
          "ci_excludes": 0.0,
          "comparator": ">",
          "effect": null,
          "metric": "err_auroc_delta_llama8b",
          "threshold": 0.03
        },
        "scoreboard_row": null,
        "statement": "the same holds on Llama-8B (cross-family)"
      }
    ],
    "id": "campaign-err-rb2",
    "kill_criteria": [
      {
        "comparator": "<=",
        "id": "K-err",
        "metric": "err_auroc_delta_min_family",
        "threshold": 0.0
      }
    ],
    "science": "S16-robustness",
    "subjects": {
      "datasets": [
        "rb2-full"
      ],
      "extra": {
        "slice_hashes": {
          "rb2-full": "ds:92765fc94e2b8231914d5119670e53bc"
        }
      },
      "organisms": [],
      "signals": [
        "skywork-v2-qwen3-8b",
        "skywork-v2-llama31-8b"
      ]
    },
    "title": "Error anticipation on RewardBench 2",
    "version": 1
  },
  "spec_hash": "spec:c8ac36e72adcc96a3d65b802efbce423",
  "study_id": "study:campaign-err-rb2@v1#c8ac36e7"
}