{
  "frozen_at": "2026-07-18T23:46:57.951556+00:00",
  "git_sha": "f93f4b5578c221d3d2665f9b305f3e86759be862+dirty",
  "spec": {
    "analysis": "campaign.judge.analyze",
    "hypotheses": [
      {
        "id": "H-vbc",
        "prediction": {
          "ci_excludes": null,
          "comparator": ">=",
          "effect": null,
          "metric": "verdict_prefix_match_rate",
          "threshold": 0.9
        },
        "scoreboard_row": null,
        "statement": "the verdict decoded at the pre-critique position matches the final verdict"
      }
    ],
    "id": "campaign-judge-vbc",
    "kill_criteria": [
      {
        "comparator": "<=",
        "id": "K-vbc",
        "metric": "verdict_prefix_match_rate",
        "threshold": 0.5
      }
    ],
    "science": "S11-values",
    "subjects": {
      "datasets": [
        "judge-pairs-1000"
      ],
      "extra": {
        "slice_hashes": {
          "judge-pairs-1000": "ds:b51b28cd7759952c0544f0d097a67ca3"
        }
      },
      "organisms": [],
      "signals": [
        "skywork-critic"
      ]
    },
    "title": "Verdict before critique on a real generative judge",
    "version": 1
  },
  "spec_hash": "spec:a8172c8c8f19fef678f524a80564dadc",
  "study_id": "study:campaign-judge-vbc@v1#a8172c8c"
}