{
  "schemaVersion": 1,
  "generatedAt": "2026-09-21T20:33:37.065Z",
  "defaultInteractiveVersion": "v1",
  "refusalInteractiveDefaultVersion": "v2",
  "newestAvailableVersion": "v2",
  "guidance": "Choose a version explicitly. V2 is newer but covers fewer models; V1 and V2 scores are not directly comparable.",
  "versions": {
    "v1": {
      "description": "Broader historical model coverage. The default interactive Overall/Capabilities view uses V1. Available benchmark scores are equally weighted; missing scores are excluded rather than treated as zero.",
      "modelHarnessRows": 23,
      "dataUpdatedAt": "2026-09-20",
      "html": "https://benchmarks.bio/results/v1/",
      "markdown": "https://benchmarks.bio/results/v1/index.md",
      "json": "https://benchmarks.bio/results/v1/latest.json",
      "csv": "https://benchmarks.bio/results/v1/latest.csv"
    },
    "v2": {
      "description": "Newer clean-v1 reruns and updated evaluation sets, currently tested on fewer frontier models. The V2 overall view requires complete coverage and uses configured evaluation-count weights. BioSecBench-Refusal defaults to V2 in the interactive site.",
      "modelHarnessRows": 9,
      "dataUpdatedAt": "2026-09-21",
      "provenance": {
        "description": "V2 score arrays in src/App.tsx, checked against the per-evaluation files in public/data/v2/. BioSecBench-Refusal scores come from public/data/v2/refusalbench.json. Overall weights come from public/data/aggregate-leaderboard.json.",
        "sourcePaths": [
          "src/App.tsx",
          "public/data/v2/",
          "public/data/aggregate-leaderboard.json"
        ]
      },
      "html": "https://benchmarks.bio/results/v2/",
      "markdown": "https://benchmarks.bio/results/v2/index.md",
      "json": "https://benchmarks.bio/results/v2/latest.json",
      "csv": "https://benchmarks.bio/results/v2/latest.csv"
    }
  }
}
