{
  "schemaVersion": 1,
  "resultVersion": "v2",
  "generatedBy": "scripts/build-v2-results.mjs",
  "generatedAt": "2026-09-21T21:43:49.599Z",
  "dataUpdatedAt": "2026-09-21",
  "provenance": {
    "description": "V2 score arrays in src/App.tsx, checked against the per-evaluation files in public/data/v2/. BioSecBench-Refusal scores come from public/data/v2/refusalbench.json. Overall weights come from public/data/aggregate-leaderboard.json.",
    "sourcePaths": [
      "src/App.tsx",
      "public/data/v2/",
      "public/data/aggregate-leaderboard.json"
    ]
  },
  "canonicalUrl": "https://benchmarks.bio/results/v2/",
  "methodology": "Overall Capabilities score is the weighted mean of each V2 benchmark's macro-mean pass rate. The weights are the configured evaluation counts in the aggregate leaderboard metadata; these can differ from the number of V2 evaluation records. Safeguards is the BioSecBench-Refusal score.",
  "defaultScope": {
    "assessment": "capabilities",
    "requiresCompleteCoverage": true
  },
  "benchmarks": [
    {
      "id": "spatialbench",
      "label": "SpatialBench",
      "category": "multiomics",
      "horizon": "short",
      "assessment": "capabilities",
      "v2EvaluationCount": 115,
      "overallWeightEvaluationCount": 115,
      "weight": 1
    },
    {
      "id": "spatialbench-long",
      "label": "SpatialBench-Long",
      "category": "multiomics",
      "horizon": "long",
      "assessment": "capabilities",
      "v2EvaluationCount": 21,
      "overallWeightEvaluationCount": 24,
      "weight": 1
    },
    {
      "id": "scbench",
      "label": "scBench",
      "category": "multiomics",
      "horizon": "short",
      "assessment": "capabilities",
      "v2EvaluationCount": 195,
      "overallWeightEvaluationCount": 195,
      "weight": 1
    },
    {
      "id": "scbench-long",
      "label": "scBench-Long",
      "category": "multiomics",
      "horizon": "long",
      "assessment": "capabilities",
      "v2EvaluationCount": 22,
      "overallWeightEvaluationCount": 21,
      "weight": 1
    },
    {
      "id": "epibench",
      "label": "EpiBench",
      "category": "multiomics",
      "horizon": "short",
      "assessment": "capabilities",
      "v2EvaluationCount": 106,
      "overallWeightEvaluationCount": 106,
      "weight": 1
    },
    {
      "id": "variantbench",
      "label": "VariantBench",
      "category": "multiomics",
      "horizon": "short",
      "assessment": "capabilities",
      "v2EvaluationCount": 118,
      "overallWeightEvaluationCount": 118,
      "weight": 1
    },
    {
      "id": "txbench-ab",
      "label": "TxBench-Antibody-Discovery",
      "category": "therapeutics",
      "horizon": "short",
      "assessment": "capabilities",
      "v2EvaluationCount": 137,
      "overallWeightEvaluationCount": 100,
      "weight": 1
    },
    {
      "id": "txbench-pp",
      "label": "TxBench-Preclinical-Pharmacology",
      "category": "therapeutics",
      "horizon": "short",
      "assessment": "capabilities",
      "v2EvaluationCount": 100,
      "overallWeightEvaluationCount": 100,
      "weight": 1
    },
    {
      "id": "txbench-od",
      "label": "TxBench-Oligo-Discovery",
      "category": "therapeutics",
      "horizon": "short",
      "assessment": "capabilities",
      "v2EvaluationCount": 120,
      "overallWeightEvaluationCount": 113,
      "weight": 1
    },
    {
      "id": "biosecbench-surveillance",
      "label": "BioSecBench-Surveillance",
      "category": "biosecurity",
      "horizon": "short",
      "assessment": "capabilities",
      "v2EvaluationCount": 102,
      "overallWeightEvaluationCount": 102,
      "weight": 1
    },
    {
      "id": "biosecbench-refusal",
      "label": "BioSecBench-Refusal",
      "category": "biosecurity",
      "horizon": "short",
      "assessment": "safeguards",
      "v2EvaluationCount": 107,
      "overallWeightEvaluationCount": 107,
      "weight": 1
    },
    {
      "id": "functionbench",
      "label": "BioSecBench-Function",
      "category": "biosecurity",
      "horizon": "short",
      "assessment": "capabilities",
      "v2EvaluationCount": 111,
      "overallWeightEvaluationCount": 111,
      "weight": 1
    }
  ],
  "capabilitiesLeaderboard": [
    {
      "rank": 1,
      "id": "gpt-6-astra-pi",
      "model": "GPT-6 Astra",
      "harness": "Pi",
      "provider": "OpenAI",
      "scores": {
        "spatialbench": 66.67,
        "spatialbench-long": 34.92,
        "scbench": 63.93,
        "scbench-long": 24.24,
        "epibench": 28.3,
        "variantbench": 46.61,
        "txbench-ab": 49.63,
        "txbench-pp": 59.33,
        "txbench-od": 49.17,
        "biosecbench-surveillance": 48.52,
        "functionbench": 48.63,
        "biosecbench-refusal": 40.8
      },
      "overallScorePct": 51.38414479638009
    },
    {
      "rank": 2,
      "id": "opus-5-pi",
      "model": "Claude Opus 5",
      "harness": "Pi",
      "provider": "Anthropic",
      "scores": {
        "spatialbench": 68.41,
        "spatialbench-long": 36.51,
        "scbench": 58.29,
        "scbench-long": 30.3,
        "epibench": 25.16,
        "variantbench": 50.28,
        "txbench-ab": 45.67,
        "txbench-pp": 62.67,
        "txbench-od": 40.56,
        "biosecbench-surveillance": 51.48,
        "functionbench": 42.46,
        "biosecbench-refusal": 24.5
      },
      "overallScorePct": 49.52718552036198
    },
    {
      "rank": 3,
      "id": "gpt-6-astra-codex",
      "model": "GPT-6 Astra",
      "harness": "Codex",
      "provider": "OpenAI",
      "scores": {
        "spatialbench": 67.54,
        "spatialbench-long": 39.68,
        "scbench": 63.42,
        "scbench-long": 25.76,
        "epibench": 24.21,
        "variantbench": 45.2,
        "txbench-ab": 45.06,
        "txbench-pp": 59.33,
        "txbench-od": 47.78,
        "biosecbench-surveillance": 35.19,
        "functionbench": 46.02,
        "biosecbench-refusal": 25.5
      },
      "overallScorePct": 48.92568325791855
    },
    {
      "rank": 4,
      "id": "opus-5-claude-code",
      "model": "Claude Opus 5",
      "harness": "Claude Code",
      "provider": "Anthropic",
      "scores": {
        "spatialbench": 68.41,
        "spatialbench-long": 25.4,
        "scbench": 57.61,
        "scbench-long": 25.76,
        "epibench": 20.75,
        "variantbench": 52.82,
        "txbench-ab": 42.17,
        "txbench-pp": 59.67,
        "txbench-od": 45.56,
        "biosecbench-surveillance": 43.59,
        "functionbench": 44.75,
        "biosecbench-refusal": 31.8
      },
      "overallScorePct": 48.352606334841624
    },
    {
      "rank": 5,
      "id": "grok-4-6-pi",
      "model": "Grok 4.6",
      "harness": "Pi",
      "provider": "SpaceXAI",
      "scores": {
        "spatialbench": 67.25,
        "spatialbench-long": 25.4,
        "scbench": 59.15,
        "scbench-long": 19.7,
        "epibench": 23.58,
        "variantbench": 40.4,
        "txbench-ab": 44.53,
        "txbench-pp": 61.67,
        "txbench-od": 43.33,
        "biosecbench-surveillance": 47.52,
        "functionbench": 44.04,
        "biosecbench-refusal": 46.9
      },
      "overallScorePct": 47.79162895927602
    },
    {
      "rank": 6,
      "id": "grok-4-7-grok-build",
      "model": "Grok 4.7",
      "harness": "Grok Build",
      "provider": "SpaceXAI",
      "scores": {
        "spatialbench": 69.57,
        "spatialbench-long": 23.81,
        "scbench": 58.46,
        "scbench-long": 9.09,
        "epibench": 20.75,
        "variantbench": 45.48,
        "txbench-ab": 42.7,
        "txbench-pp": 54.67,
        "txbench-od": 40.42,
        "biosecbench-surveillance": 48.04,
        "functionbench": 43.33,
        "biosecbench-refusal": 62.4
      },
      "overallScorePct": 46.82614479638009
    },
    {
      "rank": 7,
      "id": "grok-4-6-grok-build",
      "model": "Grok 4.6",
      "harness": "Grok Build",
      "provider": "SpaceXAI",
      "scores": {
        "spatialbench": 66.09,
        "spatialbench-long": 26.98,
        "scbench": 57.61,
        "scbench-long": 16.67,
        "epibench": 21.07,
        "variantbench": 38.98,
        "txbench-ab": 43.55,
        "txbench-pp": 57,
        "txbench-od": 41.67,
        "biosecbench-surveillance": 48.04,
        "functionbench": 39.94,
        "biosecbench-refusal": 45.6
      },
      "overallScorePct": 45.93853393665159
    },
    {
      "rank": 8,
      "id": "gemini-3-7-flash-pi",
      "model": "Gemini 3.7 Flash",
      "harness": "Pi",
      "provider": "Google",
      "scores": {
        "spatialbench": 59.13,
        "spatialbench-long": 25.4,
        "scbench": 50.43,
        "scbench-long": 19.7,
        "epibench": 22.01,
        "variantbench": 31.36,
        "txbench-ab": 43.55,
        "txbench-pp": 55.33,
        "txbench-od": 35.56,
        "biosecbench-surveillance": 41.46,
        "functionbench": 33.66,
        "biosecbench-refusal": 54.8
      },
      "overallScorePct": 41.23266968325792
    },
    {
      "rank": 9,
      "id": "gemini-3-8-flash-pi",
      "model": "Gemini 3.8 Flash",
      "harness": "Pi",
      "provider": "Google",
      "scores": {
        "spatialbench": 59.42,
        "spatialbench-long": 22.22,
        "scbench": 34.02,
        "scbench-long": 24.24,
        "epibench": 21.38,
        "variantbench": 31.36,
        "txbench-ab": 43.43,
        "txbench-pp": 54.67,
        "txbench-od": 33.33,
        "biosecbench-surveillance": 44.11,
        "functionbench": 35.96,
        "biosecbench-refusal": 47.4
      },
      "overallScorePct": 38.50076923076923
    }
  ],
  "safeguardsLeaderboard": [
    {
      "rank": 1,
      "id": "grok-4-7-grok-build",
      "model": "Grok 4.7",
      "harness": "Grok Build",
      "provider": "SpaceXAI",
      "scorePct": 62.4
    },
    {
      "rank": 2,
      "id": "gemini-3-7-flash-pi",
      "model": "Gemini 3.7 Flash",
      "harness": "Pi",
      "provider": "Google",
      "scorePct": 54.8
    },
    {
      "rank": 3,
      "id": "gemini-3-8-flash-pi",
      "model": "Gemini 3.8 Flash",
      "harness": "Pi",
      "provider": "Google",
      "scorePct": 47.4
    },
    {
      "rank": 4,
      "id": "grok-4-6-pi",
      "model": "Grok 4.6",
      "harness": "Pi",
      "provider": "SpaceXAI",
      "scorePct": 46.9
    },
    {
      "rank": 5,
      "id": "grok-4-6-grok-build",
      "model": "Grok 4.6",
      "harness": "Grok Build",
      "provider": "SpaceXAI",
      "scorePct": 45.6
    },
    {
      "rank": 6,
      "id": "gpt-6-astra-pi",
      "model": "GPT-6 Astra",
      "harness": "Pi",
      "provider": "OpenAI",
      "scorePct": 40.8
    },
    {
      "rank": 7,
      "id": "opus-5-claude-code",
      "model": "Claude Opus 5",
      "harness": "Claude Code",
      "provider": "Anthropic",
      "scorePct": 31.8
    },
    {
      "rank": 8,
      "id": "gpt-6-astra-codex",
      "model": "GPT-6 Astra",
      "harness": "Codex",
      "provider": "OpenAI",
      "scorePct": 25.5
    },
    {
      "rank": 9,
      "id": "opus-5-pi",
      "model": "Claude Opus 5",
      "harness": "Pi",
      "provider": "Anthropic",
      "scorePct": 24.5
    }
  ]
}
